diff --git a/.gitattributes b/.gitattributes index 4f2a87cb..2c87bb6f 100644 --- a/.gitattributes +++ b/.gitattributes @@ -12,3 +12,9 @@ *.cmd text eol=lf *.slnx text eol=lf *.list text eol=lf +*.glsl text eol=lf +*.vert text eol=lf +*.frag text eol=lf +*.comp text eol=lf +*.vsh text eol=lf +*.fsh text eol=lf diff --git a/.gitignore b/.gitignore index 848b2cb9..98e8d3d0 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,9 @@ baseline/ build/ .baseline/ +# Decompiled vanilla reference tree (proprietary; local read-only aid, never committed) +/_ref/ + # Anchored to repo root: reconstructed fork trees (patches stay tracked) /VintagestoryApi/ /Cairo/ @@ -63,7 +66,35 @@ graphify-out/ __pycache__/ # Private development artifacts -docs/ +# (docs/* rather than docs/, so the one tracked deliverable below can be excepted - +# a negation inside a fully excluded directory is never reconsidered by git.) +docs/* +# ...except the TAA acceptance checklist: it is a deliverable the P5 coverage test +# (Optimum.Tests/Temporal/AcceptanceTests.cs) reads, so it has to be tracked. +!docs/vulkan.md +# ...and the frozen temporal frame contract, for the same reason: the P6 stability +# test (Optimum.Tests/Temporal/ContractTests.cs) reads it. +!docs/vulkan.md +# ...and the Vulkan acceptance checklist and the parity allowlist: ssim.py reads the +# allowlist, and Optimum.Tests/parity-dump-coverage-tests.cs reads both. +!docs/vulkan.md +!docs/vulkan.md +# ...and the research notes the plan and its designs cite. +!docs/research/ +# ...and the native shader interface contract every program family is written against. +!docs/vulkan.md +# ...and the native render systems design (Phase 3b) the system stages follow. +!docs/vulkan.md +# ...and the modder documentation for Vulkan-native mod support (passes, motion writers, shaders). +!docs/vulkan.md build-linux.sh build-macos.sh build-windows.ps1 + +# Working notes for assistants and the local plan and status files: kept beside the checkout, never in it. +AGENTS.md +CLAUDE.md +.claude/ +docs/vulkan-native-plan.md +docs/vulkan-branch-progress.md +scripts/dev/harvest-maps.py diff --git a/Makefile b/Makefile index c01d97db..04096e1b 100644 --- a/Makefile +++ b/Makefile @@ -37,9 +37,9 @@ BOOTSTRAP_ARGS := ifneq ($(CLIENT_ARCHIVE),) BOOTSTRAP_ARGS += --client-archive $(CLIENT_ARCHIVE) endif -BOOTSTRAP_ARGS := --version $(VERSION) +BOOTSTRAP_ARGS += --version $(VERSION) -.PHONY: help check check-patches check-compat check-shaders bootstrap bootstrap-git-test build clean refresh patches patch-il deploy run run-creative run-connect \ +.PHONY: help check check-patches check-compat check-shaders check-shaders-vk bootstrap bootstrap-git-test build clean refresh patches patch-il deploy run run-creative run-connect \ package package-overlay package-linux package-appimage package-macos package-win bench-scaling worldgen-benchmark-test worldgen-benchmark-smoke worldgen-benchmark \ coverage mutate-launcher server-smoke @@ -58,6 +58,12 @@ check-compat: ## Verify patches keep vanilla multiplayer compatibility guards check-shaders: ## Verify optimized shader overlays are not truncated bash scripts/validate-shader-assets.sh sources/shaders +SHADER_COMPILER = tools/shader-compiler/bin/$(CONFIGURATION)/net10.0/Optimum.Shaders.Compiler.dll + +check-shaders-vk: ## Recompile sources/shaders-vk and fail on any SPIR-V or manifest difference from the build output + dotnet build tools/shader-compiler/Optimum.Shaders.Compiler.csproj -c $(CONFIGURATION) --nologo -v quiet -p:OptimumSkipNativeShaders=true + dotnet $(SHADER_COMPILER) --verify sources/shaders-vk $(MOD_OUT) + bootstrap: ## Download client, decompile, clone forks, apply patches bash scripts/bootstrap.sh $(BOOTSTRAP_ARGS) @@ -102,7 +108,39 @@ deploy: patch-il check-shaders ## Deploy Cecil-patched DLLs into vanilla client @cp $(MOD_OUT)/VSSurvivalMod.dll $(VANILLA_DIR)/Mods/ @cp $(MOD_OUT)/VSCreativeMod.dll $(VANILLA_DIR)/Mods/ @cp $(MOD_OUT)/cairo-sharp.dll $(VANILLA_DIR)/Lib/ - @cp sources/shaders/*.fsh sources/shaders/*.vsh $(VANILLA_DIR)/assets/game/shaders/ + @# The Vulkan renderer and its dependencies, loaded by name at startup; a + @# stale copy here makes the probe throw and the client fall back to OpenGL. + @cp $(MOD_OUT)/Optimum.Render.Vulkan.dll $(VANILLA_DIR)/ + @cp $(MOD_OUT)/Silk.NET.*.dll $(VANILLA_DIR)/ + @# shaderc goes into the application root: Silk.NET.Shaderc probes the application + @# directory and LD_LIBRARY_PATH, not Lib/, and a copy it cannot find makes the + @# renderer fall back to OpenGL silently. + @if [ -f "$(MOD_OUT)/runtimes/linux-x64/native/libshaderc_shared.so" ]; then cp $(MOD_OUT)/runtimes/linux-x64/native/libshaderc_shared.so $(VANILLA_DIR)/; fi + @# Native SPIR-V and its manifest (docs/vulkan.md), beside the + @# renderer and never under assets/: the asset manager must not read SPIR-V and a mod must + @# not shadow engine shaders by asset priority. The build always writes the manifest, even + @# for an empty source tree, so a missing one means tools/shader-compiler never ran. The + @# directory is replaced whole, so a removed program does not linger, and checked file by + @# file like the overlays below. + @[ -f "$(MOD_OUT)/shaders-vk/shaders.manifest.json" ] || { echo "Error: $(MOD_OUT)/shaders-vk/shaders.manifest.json missing; build tools/shader-compiler (dotnet build VintageStory.slnx)"; exit 1; } + @rm -rf "$(VANILLA_DIR)/shaders-vk" && mkdir -p "$(VANILLA_DIR)/shaders-vk" && cp -f $(MOD_OUT)/shaders-vk/* "$(VANILLA_DIR)/shaders-vk/" + @for f in $(MOD_OUT)/shaders-vk/*; do d="$(VANILLA_DIR)/shaders-vk/$$(basename $$f)"; cmp -s "$$f" "$$d" || { echo "Error: $$f did not reach $$d (missing or content differs)"; exit 1; }; done + @# Every file, not *.fsh plus *.vsh: the packagers copy the whole directory, + @# and a stage that ships only on one of the two paths is the bug the + @# completeness check below exists to catch. + @for f in sources/shaders/*; do [ -f "$$f" ] || continue; cp -f "$$f" "$(VANILLA_DIR)/assets/game/shaders/$$(basename $$f)" || exit 1; done + @# Shader includes (TAA P3: the WarpState vertexwarp.vsh). Same override + @# mechanism as shaders - ShaderRegistry merges both asset categories into one + @# include dictionary - but a separate directory, so it needs its own copy. + @if [ -d "sources/shaderincludes" ]; then mkdir -p $(VANILLA_DIR)/assets/game/shaderincludes; for f in sources/shaderincludes/*; do [ -f "$$f" ] || continue; cp -f "$$f" "$(VANILLA_DIR)/assets/game/shaderincludes/$$(basename $$f)" || exit 1; done; fi + @# Both copies above are wildcards, so a file that never arrives means a moved + @# source path or a missing destination directory, not a forgotten list entry - + @# and the symptom is silent, vanilla's shader running in place of Optimum's. + @# TAA fails worst that way: its stages (taa-resolve, taa-debug, taa-skymotion, + @# taa-sharpen), the liquid velocity pass and the includes the motion writers + @# compile against have to arrive together or the resolve reads vectors nobody + @# wrote. Fail the deploy instead. + @for f in sources/shaders/* sources/shaderincludes/*; do [ -f "$$f" ] || continue; d="$(VANILLA_DIR)/assets/game/$$(echo $$f | cut -d/ -f2)/$$(basename $$f)"; cmp -s "$$f" "$$d" || { echo "Error: $$f did not reach $$d (missing or content differs)"; exit 1; }; done @if [ -d "sources/lang" ]; then for f in sources/lang/*.json; do [ -f "$$f" ] || continue; dst="$(VANILLA_DIR)/assets/game/lang/$$(basename $$f)"; [ -f "$$dst" ] || continue; python3 -c "import json,sys; s=json.load(open(sys.argv[1],encoding='utf-8-sig')); d=json.load(open(sys.argv[2],encoding='utf-8-sig')); d.update(s); json.dump(d,open(sys.argv[2],'w',encoding='utf-8'),ensure_ascii=False,indent='\t')" "$$f" "$$dst"; done; fi @if [ -d "$(INSTALL_DIR)" ]; then \ echo "Deploying to $(INSTALL_DIR)..."; \ @@ -115,7 +153,13 @@ deploy: patch-il check-shaders ## Deploy Cecil-patched DLLs into vanilla client cp $(MOD_OUT)/VSSurvivalMod.dll $(INSTALL_DIR)/Mods/; \ cp $(MOD_OUT)/VSCreativeMod.dll $(INSTALL_DIR)/Mods/; \ cp $(MOD_OUT)/cairo-sharp.dll $(INSTALL_DIR)/Lib/; \ - cp sources/shaders/*.fsh sources/shaders/*.vsh $(INSTALL_DIR)/assets/game/shaders/; \ + cp $(MOD_OUT)/Optimum.Render.Vulkan.dll $(INSTALL_DIR)/; cp $(MOD_OUT)/Silk.NET.*.dll $(INSTALL_DIR)/; \ + if [ -f "$(MOD_OUT)/runtimes/linux-x64/native/libshaderc_shared.so" ]; then cp $(MOD_OUT)/runtimes/linux-x64/native/libshaderc_shared.so $(INSTALL_DIR)/; fi; \ + rm -rf "$(INSTALL_DIR)/shaders-vk" && mkdir -p "$(INSTALL_DIR)/shaders-vk" && cp -f $(MOD_OUT)/shaders-vk/* "$(INSTALL_DIR)/shaders-vk/" || exit 1; \ + for f in $(MOD_OUT)/shaders-vk/*; do d="$(INSTALL_DIR)/shaders-vk/$$(basename $$f)"; cmp -s "$$f" "$$d" || { echo "Error: $$f did not reach $$d (missing or content differs)"; exit 1; }; done; \ + for f in sources/shaders/*; do [ -f "$$f" ] || continue; cp -f "$$f" "$(INSTALL_DIR)/assets/game/shaders/$$(basename $$f)" || exit 1; done; \ + if [ -d "sources/shaderincludes" ]; then mkdir -p $(INSTALL_DIR)/assets/game/shaderincludes; for f in sources/shaderincludes/*; do [ -f "$$f" ] || continue; cp -f "$$f" "$(INSTALL_DIR)/assets/game/shaderincludes/$$(basename $$f)" || exit 1; done; fi; \ + for f in sources/shaders/* sources/shaderincludes/*; do [ -f "$$f" ] || continue; d="$(INSTALL_DIR)/assets/game/$$(echo $$f | cut -d/ -f2)/$$(basename $$f)"; cmp -s "$$f" "$$d" || { echo "Error: $$f did not reach $$d (missing or content differs)"; exit 1; }; done; \ if [ -d "sources/lang" ]; then for f in sources/lang/*.json; do [ -f "$$f" ] || continue; dst="$(INSTALL_DIR)/assets/game/lang/$$(basename $$f)"; [ -f "$$dst" ] || continue; python3 -c "import json,sys; s=json.load(open(sys.argv[1],encoding='utf-8-sig')); d=json.load(open(sys.argv[2],encoding='utf-8-sig')); d.update(s); json.dump(d,open(sys.argv[2],'w',encoding='utf-8'),ensure_ascii=False,indent='\t')" "$$f" "$$dst"; done; fi; \ fi @echo "Deploy complete." diff --git a/Optimum.Launcher.Tests/ShaderCompatibilityScannerTests.cs b/Optimum.Launcher.Tests/ShaderCompatibilityScannerTests.cs index 8fe5b972..03cffc02 100644 --- a/Optimum.Launcher.Tests/ShaderCompatibilityScannerTests.cs +++ b/Optimum.Launcher.Tests/ShaderCompatibilityScannerTests.cs @@ -1,6 +1,7 @@ using System; using System.IO; using System.IO.Compression; +using System.Linq; using System.Text; using Optimum.Launcher; using Vintagestory.API.Config; @@ -117,6 +118,305 @@ public void SavedShaderCompatibilityReportDisablesEffectiveMapPageCache() } } + [Theory] + // Optimum's own temporal stages: an external copy of any of them is not a + // writer that stops emitting, it is a second resolve compiled against an MRT + // layout, history format and reactive convention it cannot know. + [InlineData("assets/mymodshaders/shaders/taa-resolve.fsh")] + [InlineData("assets/mymodshaders/shaders/taa-debug.vsh")] + [InlineData("assets/mymodshaders/shaders/taa-skymotion.fsh")] + [InlineData("assets/mymodshaders/shaders/taa-sharpen.fsh")] + // A stage Optimum has not written yet: the "taa-" prefix rule has to cover + // it, or every future stage ships without a scanner verdict. + [InlineData("assets/mymodshaders/shaders/taa-somethingnew.fsh")] + // The liquid velocity pass and the FSR pair the post-resolve sharpen shares + // its vertex stage and lobe maths with. + [InlineData("assets/mymodshaders/shaders/chunkliquidmotion.vsh")] + [InlineData("assets/mymodshaders/shaders/fsr-rcas.fsh")] + [InlineData("assets/mymodshaders/shaders/fsr-easu.vsh")] + // ShaderRegistry merges every shaderinclude into one dictionary that all the + // motion writers compile against, so any file in that directory can redefine + // a helper they call - not only the vertexwarp include named in the rules. + [InlineData("assets/mymodshaders/shaderincludes/vertexwarp.vsh")] + [InlineData("assets/mymodshaders/shaderincludes/somehelper.vsh")] + // Vanilla's shaderincludes/ also carries .ash files, and ShaderRegistry loads + // every include regardless of extension - so the stage-extension filter must + // not apply here or an external .ash helper is invisible to the scanner. + [InlineData("assets/mymodshaders/shaderincludes/foo.ash")] + [InlineData("assets/mymodshaders/shaderincludes/vertexflagbits.ash")] + public void AnExternalCopyOfAnyTaaShaderDisablesTaa(string entryPath) + { + string dataPath = Path.Combine(_root, "data"); + string archivePath = Path.Combine(dataPath, "Mods", "SomeShaderPack.zip"); + Directory.CreateDirectory(Path.GetDirectoryName(archivePath)!); + using (ZipArchive archive = ZipFile.Open(archivePath, ZipArchiveMode.Create)) + { + using StreamWriter writer = new(archive.CreateEntry(entryPath).Open()); + writer.Write("void main() { }"); + } + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan( + dataPath, + Path.Combine(_root, "game"), + "test"); + + Assert.Contains("Taa", report.DisabledFeatures); + Assert.Contains( + "external shader owns a motion-vector writer contract", + report.FeatureReasons["Taa"]); + } + + [Theory] + [InlineData("assets/mymodshaders/shaders/gui.fsh")] + // Under shaders/ a non-stage extension is not a shader at all: the relaxed + // extension rule belongs to shaderincludes/ only. + [InlineData("assets/mymodshaders/shaders/readme.ash")] + public void AnUnrelatedExternalShaderLeavesTaaAlone(string entryPath) + { + string dataPath = Path.Combine(_root, "data"); + string archivePath = Path.Combine(dataPath, "Mods", "SomeShaderPack.zip"); + Directory.CreateDirectory(Path.GetDirectoryName(archivePath)!); + using (ZipArchive archive = ZipFile.Open(archivePath, ZipArchiveMode.Create)) + { + using StreamWriter writer = new(archive.CreateEntry(entryPath).Open()); + writer.Write("void main() { }"); + } + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan( + dataPath, + Path.Combine(_root, "game"), + "test"); + + // The veto is explicit, not a blanket "some mod ships shaders" reaction: + // TAA stays available and only an owned contract turns it off. + Assert.DoesNotContain("Taa", report.DisabledFeatures); + } + + // ------------------------------------------------ v2: vanilla-shader overrides + + [Fact] + public void AModOverridingChunkopaqueMarksOnlyThatProgramForTheRewriter() + { + string dataPath = Path.Combine(_root, "data"); + WriteArchive(Path.Combine(dataPath, "Mods", "OpaquePack.zip"), + "assets/opaquepack/shaders/chunkopaque.fsh", + // A new program name is the mod's own: it has no native blob to bypass. + "assets/opaquepack/shaders/opaquepack-glow.fsh"); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + Assert.Equal(["chunkopaque"], report.RewriterPrograms); + ShaderAssetOverride finding = Assert.Single(report.ShaderAssetOverrides); + // NormalizeShaderPath keeps only "shaders/..." for a domain other than game. + Assert.Equal("shaders/chunkopaque.fsh", finding.Asset); + Assert.Equal(["chunkopaque"], finding.Programs); + Assert.Single(finding.Owners); + Assert.False(report.OpenGlRequired); + } + + [Fact] + public void AShaderincludesOverrideMarksEveryProgram() + { + string dataPath = Path.Combine(_root, "data"); + WriteArchive(Path.Combine(dataPath, "Mods", "FogPack.zip"), + "assets/fogpack/shaderincludes/fogandlight.fsh", + "assets/fogpack/shaders/sky.fsh"); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + Assert.Equal([ShaderCompatibilityScanner.AllPrograms], report.RewriterPrograms); + Assert.Contains(report.ShaderAssetOverrides, x => + x.Asset == "shaderincludes/fogandlight.fsh" || x.Asset.EndsWith("shaderincludes/fogandlight.fsh", StringComparison.Ordinal)); + Assert.Contains(report.ShaderAssetOverrides, x => x.Programs.SequenceEqual([ShaderCompatibilityScanner.AllPrograms])); + Assert.Contains(report.ShaderAssetOverrides, x => x.Programs.SequenceEqual(["sky"])); + Assert.False(report.OpenGlRequired); + } + + [Fact] + public void AProgramOnlyTheInstalledGameShipsIsStillAnOverride() + { + string gameDir = Path.Combine(_root, "game"); + Directory.CreateDirectory(Path.Combine(gameDir, "assets", "game", "shaders")); + File.WriteAllText(Path.Combine(gameDir, "assets", "game", "shaders", "futureprogram.vsh"), "void main() { }"); + string dataPath = Path.Combine(_root, "data"); + WriteArchive(Path.Combine(dataPath, "Mods", "FuturePack.zip"), "assets/futurepack/shaders/futureprogram.vsh"); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, gameDir, "test"); + + Assert.Equal(["futureprogram"], report.RewriterPrograms); + } + + [Fact] + public void AFailedScanSendsEveryProgramToTheRewriterButDoesNotVetoVulkan() + { + ShaderCompatibilityReport report = ShaderCompatibilityScanner.CreateConservativeReport("test", "boom"); + + Assert.Equal([ShaderCompatibilityScanner.AllPrograms], report.RewriterPrograms); + Assert.False(report.OpenGlRequired); + } + + /// + /// Every native program is one a mod can override by file name; a program missing + /// from the scanner's list would link its SPIR-V over a mod's source whenever the + /// installed game directory is not readable. + /// + [Fact] + public void EveryNativeProgramIsAKnownOverridableProgram() + { + string shadersVk = Path.Combine(FindRepositoryRoot(), "sources", "shaders-vk"); + string[] native = Directory.EnumerateFiles(shadersVk, "*.glsl", SearchOption.TopDirectoryOnly) + .Select(Path.GetFileNameWithoutExtension) + .Cast() + .ToArray(); + + Assert.NotEmpty(native); + Assert.Empty(native.Except(ShaderCompatibilityScanner.KnownPrograms, StringComparer.OrdinalIgnoreCase)); + } + + // ------------------------------------------------ v2: PlatformInternals + + [Fact] + public void HarmonyPlusAClientPlatformWindowsNameRoutesToOpenGl() + { + string dataPath = Path.Combine(_root, "data"); + string modsPath = Path.Combine(dataPath, "Mods"); + Directory.CreateDirectory(modsPath); + File.WriteAllBytes(Path.Combine(modsPath, "PlatformPatch.dll"), + Encoding.UTF8.GetBytes("0Harmony HarmonyLib HarmonyPatch ClientPlatformWindows RenderMesh")); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + ShaderModSource source = Assert.Single(report.Sources); + Assert.Contains("PlatformInternals", source.Indicators); + Assert.DoesNotContain("RawOpenGL", source.Indicators); + Assert.True(report.OpenGlRequired); + Assert.Contains("Vulkan", report.DisabledFeatures); + Assert.Equal([source.Id], report.OpenGlRequiredBy); + Assert.Contains(report.FeatureReasons["Vulkan"], reason => reason.Contains("Harmony", StringComparison.Ordinal)); + } + + [Fact] + public void AUtf16UserStringNamingShaderProgramBaseCountsAsAPlatformReference() + { + string dataPath = Path.Combine(_root, "data"); + string modsPath = Path.Combine(dataPath, "Mods"); + Directory.CreateDirectory(modsPath); + byte[] metadata = Encoding.UTF8.GetBytes("0Harmony AccessTools "); + // An odd prefix puts the user string at an odd offset, as the #US heap may. + byte[] userString = [0x2B, .. Encoding.Unicode.GetBytes("Vintagestory.Client.NoObf.ShaderProgramBase:Use")]; + File.WriteAllBytes(Path.Combine(modsPath, "ReflectionPatch.dll"), [.. metadata, .. userString]); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + Assert.Contains("PlatformInternals", Assert.Single(report.Sources).Indicators); + Assert.True(report.OpenGlRequired); + } + + [Theory] + // A platform name without Harmony is a reflection read, not a patch. + // (RegisterRenderer keeps the assembly a listed source.) + [InlineData("ClientPlatformWindows ScreenshotHelper RegisterRenderer")] + // Harmony on gameplay code does not touch the graphics seam. + [InlineData("0Harmony HarmonyLib EntityBehaviorHealth OnDamage")] + public void PlatformNamesOrHarmonyAloneStayOnVulkan(string content) + { + string dataPath = Path.Combine(_root, "data"); + string modsPath = Path.Combine(dataPath, "Mods"); + Directory.CreateDirectory(modsPath); + File.WriteAllBytes(Path.Combine(modsPath, "Gameplay.dll"), Encoding.UTF8.GetBytes(content)); + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + Assert.DoesNotContain("PlatformInternals", Assert.Single(report.Sources).Indicators); + Assert.False(report.OpenGlRequired); + } + + [Fact] + public void ACleanModStaysOnVulkanWithNativePrograms() + { + string dataPath = Path.Combine(_root, "data"); + string archivePath = Path.Combine(dataPath, "Mods", "CleanMod.zip"); + Directory.CreateDirectory(Path.GetDirectoryName(archivePath)!); + using (ZipArchive archive = ZipFile.Open(archivePath, ZipArchiveMode.Create)) + { + using (StreamWriter info = new(archive.CreateEntry("modinfo.json").Open())) + info.Write("{\"modid\":\"cleanmod\",\"name\":\"Clean Mod\",\"version\":\"1.0.0\"}"); + using (Stream dll = archive.CreateEntry("CleanMod.dll").Open()) + dll.Write(Encoding.UTF8.GetBytes("Vintagestory.API.Common ModSystem ICoreClientAPI BlockEntity")); + } + + ShaderCompatibilityReport report = ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test"); + + Assert.Equal("cleanmod", Assert.Single(report.Sources).Id); + Assert.False(report.OpenGlRequired); + Assert.Empty(report.OpenGlRequiredBy); + Assert.Empty(report.RewriterPrograms); + Assert.Empty(report.ShaderAssetOverrides); + Assert.DoesNotContain("Vulkan", report.DisabledFeatures); + } + + // ------------------------------------------------ v2: schema + + [Fact] + public void AVersion1ReportIsInvalidatedAndTheNextScanReplacesIt() + { + string dataPath = Path.Combine(_root, "data"); + string reportPath = Path.Combine(dataPath, ".optimum", ShaderCompatibilityScanner.ReportFileName); + Directory.CreateDirectory(Path.GetDirectoryName(reportPath)!); + File.WriteAllText(reportPath, + "{\"schemaVersion\":1,\"optimumVersion\":\"old\",\"scanFailed\":false,\"disabledFeatures\":[]}"); + + Assert.Equal(2, ShaderCompatibilityScanner.CurrentSchemaVersion); + Assert.Null(ShaderCompatibilityScanner.LoadReport(dataPath)); + + WriteArchive(Path.Combine(dataPath, "Mods", "OpaquePack.zip"), "assets/opaquepack/shaders/chunkopaque.vsh"); + ShaderCompatibilityScanner.SaveReport(dataPath, + ShaderCompatibilityScanner.Scan(dataPath, Path.Combine(_root, "game"), "test")); + + ShaderCompatibilityReport? loaded = ShaderCompatibilityScanner.LoadReport(dataPath); + Assert.NotNull(loaded); + Assert.Equal(ShaderCompatibilityScanner.CurrentSchemaVersion, loaded.SchemaVersion); + Assert.Equal(["chunkopaque"], loaded.RewriterPrograms); + Assert.Contains("\"rewriterPrograms\"", File.ReadAllText(reportPath), StringComparison.Ordinal); + } + + [Theory] + [InlineData("{\"optimumVersion\":\"old\"}")] + [InlineData("{\"schemaVersion\":\"2\"}")] + [InlineData("{\"schemaVersion\":3}")] + [InlineData("not json")] + public void AReportWithoutTheCurrentSchemaVersionIsNotLoaded(string content) + { + string dataPath = Path.Combine(_root, "data"); + string reportPath = Path.Combine(dataPath, ".optimum", ShaderCompatibilityScanner.ReportFileName); + Directory.CreateDirectory(Path.GetDirectoryName(reportPath)!); + File.WriteAllText(reportPath, content); + + Assert.Null(ShaderCompatibilityScanner.LoadReport(dataPath)); + } + + private static void WriteArchive(string archivePath, params string[] entries) + { + Directory.CreateDirectory(Path.GetDirectoryName(archivePath)!); + using ZipArchive archive = ZipFile.Open(archivePath, ZipArchiveMode.Create); + foreach (string entry in entries) + { + using StreamWriter writer = new(archive.CreateEntry(entry).Open()); + writer.Write("void main() { }"); + } + } + + private static string FindRepositoryRoot() + { + var directory = new DirectoryInfo(AppContext.BaseDirectory); + while (directory is not null) + { + if (File.Exists(Path.Combine(directory.FullName, "VintageStory.slnx"))) return directory.FullName; + directory = directory.Parent; + } + throw new DirectoryNotFoundException("Repository root with VintageStory.slnx not found."); + } + private static void WriteHookAssembly(string path) { File.WriteAllBytes(path, Encoding.UTF8.GetBytes( diff --git a/Optimum.Launcher/ShaderCompatibilityScanner.cs b/Optimum.Launcher/ShaderCompatibilityScanner.cs index 901ebe20..0e649e32 100644 --- a/Optimum.Launcher/ShaderCompatibilityScanner.cs +++ b/Optimum.Launcher/ShaderCompatibilityScanner.cs @@ -16,9 +16,61 @@ namespace Optimum.Launcher; /// public static class ShaderCompatibilityScanner { - public const int CurrentSchemaVersion = 1; + /// + /// 2 (2026-09-15): adds , + /// , the PlatformInternals + /// indicator and . The launcher + /// rescans and overwrites the report on every start; + /// refuses any other version, so a v1 file is never read as if it carried the v2 + /// verdicts. + /// + public const int CurrentSchemaVersion = 2; public const string ReportFileName = "shader-compatibility.json"; + /// + /// The single entry holds + /// when every program has to take the rewriter: a shaderincludes override, or a + /// report the scanner could not complete. + /// + public const string AllPrograms = "all"; + + /// + /// Program base names the client registers from assets/game/shaders + /// (vanilla plus the Optimum overrides and stages that ship there). A mod file + /// assets/<domain>/shaders/<name>.vsh|.fsh|.gsh whose stem is one of + /// these replaces that program's source, so the Vulkan path must build it through + /// the rewriter from the mod's GLSL instead of linking the native SPIR-V. The scan + /// also adds every stem it finds in the installed game's own shaders directory. + /// + private static readonly string[] BuiltInPrograms = + [ + "aurora", "autocamera", "bilateralblur", "blit", "blockhighlights", "blur", "celestialobject", + "chunkliquid", "chunkliquiddepth", "chunkliquidmotion", "chunkopaque", "chunkshadowmap", "chunktopsoil", + "chunktransparent", "cloudmap", "clouds", "cloudvolumetric", "colorgrade", "debugdepthbuffer", "decals", + "entityanimated", "final", "findbright", "fsr-easu", "fsr-rcas", "godrays", "gui", "guigear", "guitopsoil", + "helditem", "instanced", "lines", "luma", "nightsky", "optimum-map", "particlescube", "particlesquad", + "particlesquad2d", "scene-ssao", "shadowmapentityanimated", "sky", "ssao", "standard", "taa-debug", + "taa-resolve", "taa-sharpen", "taa-skymotion", "texture2texture", "transparentcompose", "ui-compose", + "upscale-ssao", "wireframe", "woittest" + ]; + + /// Program base names the scanner treats as overridable without an installed game. + public static IReadOnlyList KnownPrograms => BuiltInPrograms; + + /// + /// Platform graphics types a Harmony patch can target. The Vulkan backend replaces + /// the platform (VulkanClientPlatform overrides the graphics members) and + /// links programs itself, so a prefix/postfix on one of these members either never + /// runs or runs against state the Vulkan path does not use. + /// + private static readonly string[] PlatformInternalTypes = + [ + "ClientPlatformWindows", "ShaderProgramBase" + ]; + + private static readonly byte[][] PlatformInternalTypesUtf16 = + [.. PlatformInternalTypes.Select(Encoding.Unicode.GetBytes)]; + private static readonly string[] ShaderFeatures = [ "GreedyMesh", "RenderScale", "GodRaysSampleCap", "EntityLightBatch", @@ -33,7 +85,8 @@ private static readonly (string Token, string Name)[] IndicatorTokens = ("harmony", "Harmony"), ("registerrenderer", "RenderHook"), ("onrenderframe", "RenderHook"), - ("enumrenderstage", "RenderHook") + ("enumrenderstage", "RenderHook"), + ("opentk.graphics", "RawOpenGL") ]; private static readonly string[] OptimumBuiltInMods = @@ -84,10 +137,64 @@ public static ShaderCompatibilityReport Scan(string dataPath, string gameDir, st MarkFailed(report, "Mods", ex); } - FinalizeReport(report); + FinalizeReport(report, InstalledPrograms(gameDir)); return report; } + /// + /// Reads a saved report, or null when there is none, it cannot be parsed, or it was + /// written under another schema version. A v1 report lacks the v2 verdicts + /// (rewriter programs, PlatformInternals routing), so it is invalidated rather than + /// migrated: its absence of a verdict must not read as "nothing found". + /// + public static ShaderCompatibilityReport? LoadReport(string dataPath) + { + string path = Path.Combine(dataPath, ".optimum", ReportFileName); + try + { + if (!File.Exists(path)) return null; + using FileStream stream = File.OpenRead(path); + using JsonDocument document = JsonDocument.Parse(stream); + if (!document.RootElement.TryGetProperty("schemaVersion", out JsonElement version) || + version.ValueKind != JsonValueKind.Number || + !version.TryGetInt32(out int schemaVersion) || + schemaVersion != CurrentSchemaVersion) + { + return null; + } + return document.RootElement.Deserialize(JsonOptions); + } + catch (Exception ex) when (IsRecoverable(ex) || ex is JsonException) + { + return null; + } + } + + private static HashSet InstalledPrograms(string gameDir) + { + var programs = new HashSet(BuiltInPrograms, StringComparer.OrdinalIgnoreCase); + try + { + string shaders = Path.Combine(gameDir, "assets", "game", "shaders"); + if (!Directory.Exists(shaders)) return programs; + foreach (string file in Directory.EnumerateFiles(shaders)) + { + if (IsStageExtension(Path.GetExtension(file))) + programs.Add(Path.GetFileNameWithoutExtension(file).ToLowerInvariant()); + } + } + catch (Exception ex) when (IsRecoverable(ex)) + { + // The built-in list still covers every program this build ships. + } + return programs; + } + + private static bool IsStageExtension(string extension) => + extension.Equals(".fsh", StringComparison.OrdinalIgnoreCase) || + extension.Equals(".vsh", StringComparison.OrdinalIgnoreCase) || + extension.Equals(".gsh", StringComparison.OrdinalIgnoreCase); + public static string SaveReport(string dataPath, ShaderCompatibilityReport report) { ArgumentNullException.ThrowIfNull(report); @@ -128,7 +235,7 @@ public static ShaderCompatibilityReport CreateConservativeReport(string optimumV report.DisabledFeatures.Add(feature); report.FeatureReasons[feature] = ["scanner failure: conservative fallback"]; } - FinalizeReport(report); + FinalizeReport(report, null); return report; } @@ -209,7 +316,7 @@ private static void ScanFile(string file, string relativePath, HashSet s byte[] bytes = File.ReadAllBytes(file); RecordContentHash(source, normalized, bytes); if (isModInfo) ReadModInfo(ReadTextBytes(bytes), source); - if (isAssembly) AddIndicators(ReadTextBytes(bytes), indicators); + if (isAssembly) AddIndicators(bytes, indicators); } private static void ScanArchive(string archivePath, HashSet shaders, HashSet indicators, ShaderModSource source, ShaderCompatibilityReport report) @@ -233,7 +340,7 @@ private static void ScanArchive(string archivePath, HashSet shaders, Has byte[] bytes = memory.ToArray(); RecordContentHash(source, relativePath, bytes); if (isModInfo) ReadModInfo(ReadTextBytes(bytes), source); - if (isAssembly) AddIndicators(ReadTextBytes(bytes), indicators); + if (isAssembly) AddIndicators(bytes, indicators); } } catch (Exception ex) when (IsRecoverable(ex)) @@ -259,17 +366,45 @@ private static void ReadModInfo(string json, ShaderModSource source) } } - private static void AddIndicators(string text, HashSet indicators) + private static void AddIndicators(byte[] bytes, HashSet indicators) { - string lower = text.ToLowerInvariant(); + string lower = ReadTextBytes(bytes).ToLowerInvariant(); foreach ((string token, string name) in IndicatorTokens) { if (lower.Contains(token, StringComparison.Ordinal)) indicators.Add(name); } + + // PlatformInternals: the assembly references Harmony and names a platform + // graphics type. Type and member references, and the type names serialized + // into [HarmonyPatch(typeof(...))] attribute blobs, are UTF-8 in metadata; + // AccessTools.Method("...ClientPlatformWindows:RenderMesh") is a user string, + // which the #US heap stores as UTF-16. Both forms count. The other indicators + // keep their UTF-8-only match so their verdicts do not move. + if (!indicators.Contains("Harmony")) return; + bool namesPlatform = + PlatformInternalTypes.Any(type => lower.Contains(type.ToLowerInvariant(), StringComparison.Ordinal)) || + PlatformInternalTypesUtf16.Any(pattern => bytes.AsSpan().IndexOf(pattern) >= 0); + if (namesPlatform) indicators.Add("PlatformInternals"); } - private static void FinalizeReport(ShaderCompatibilityReport report) + private static void FinalizeReport(ShaderCompatibilityReport report, HashSet? installedPrograms) { + AddShaderAssetOverrides(report, installedPrograms); + + // A Harmony patch on a platform graphics member is not honoured on Vulkan: + // VulkanClientPlatform overrides those members and links programs itself, + // so the patched GL body never runs. Like RawOpenGL this routes the session + // to OpenGL, and like it the decision is not a ShaderFeature, so a scan + // failure alone never forces it. + report.OpenGlRequiredBy = report.Sources + .Where(x => x.Indicators.Contains("RawOpenGL") || x.Indicators.Contains("PlatformInternals")) + .Select(x => x.Id) + .Distinct(StringComparer.OrdinalIgnoreCase) + .Order(StringComparer.OrdinalIgnoreCase) + .ToList(); + AddFeatureDecision(report, "Vulkan", report.Sources.Any(x => x.Indicators.Contains("PlatformInternals")), + "a mod Harmony-patches a platform graphics member (ClientPlatformWindows or ShaderProgramBase), which the Vulkan backend does not honour"); + foreach (List owners in report.ShaderOwners.Values) owners.Sort(StringComparer.OrdinalIgnoreCase); @@ -309,10 +444,85 @@ private static void FinalizeReport(ShaderCompatibilityReport report) AddFeatureDecision(report, "MapPageCache", externalShaderAssets || externalShaderHooks, "external shader assets or shader hooks can dispose registered programs during reload"); + // "Vulkan" is a backend decision, not a shader feature, and is deliberately + // absent from ShaderFeatures: a scanner failure must not silently veto a + // backend the user explicitly asked for. GLSL that mods ship is fine - the + // translator compiles arbitrary sources - but a mod issuing GL calls itself + // has no path through a Vulkan device. + bool rawOpenGl = report.Sources.Any(x => x.Indicators.Contains("RawOpenGL")); + AddFeatureDecision(report, "Vulkan", rawOpenGl, + "a mod calls OpenGL directly, which the Vulkan backend cannot serve"); + + // "Taa" is deliberately absent from ShaderFeatures for the same reason as + // "Vulkan": OptimumConfig.EffectiveTaa consults IsFeatureExplicitlyDisabled, + // so a missing scan must not silently veto a renderer feature the user + // asked for - only an explicit verdict does. + // + // An external copy of any shader Optimum's motion-vector writers live in, + // or of the vertexwarp include they evaluate twice, replaces the writer + // with one that emits nothing to the motion attachment. The resolve would + // then reproject those pixels by camera motion alone while everything + // around them used real vectors, which is worse than not running TAA. + bool externalMotionShader = + HasExternalShader(report, "chunkopaque.vsh") || HasExternalShader(report, "chunkopaque.fsh") || + HasExternalShader(report, "chunktopsoil.vsh") || HasExternalShader(report, "chunktopsoil.fsh") || + HasExternalShader(report, "entityanimated.vsh") || HasExternalShader(report, "entityanimated.fsh") || + HasExternalShader(report, "standard.vsh") || HasExternalShader(report, "standard.fsh") || + HasExternalShader(report, "instanced.vsh") || HasExternalShader(report, "instanced.fsh") || + // The liquid velocity pass re-draws the liquid pools through its own + // program and has to land on exactly the surface chunkliquid.vsh + // shaded; an external copy of either file breaks that agreement, and + // its vectors would then be rejected or - worse - accepted for a + // surface half a pixel away. + HasExternalShader(report, "chunkliquid.vsh") || + HasExternalShader(report, "chunkliquidmotion.vsh") || HasExternalShader(report, "chunkliquidmotion.fsh") || + // Cube particles write the motion attachment themselves, and the + // OIT merge is where every transparent that cannot write it gets its + // reactive value. An external copy of either drops that content back + // to camera reprojection with no reactive flag at all, which ghosts + // exactly the fast-moving, alpha-blended pixels TAA is worst at. + HasExternalShader(report, "particlescube.vsh") || HasExternalShader(report, "particlescube.fsh") || + HasExternalShader(report, "transparentcompose.fsh") || + // A decal that no longer writes the attachment leaves the block's + // vector behind a depth the decal itself moved, which the resolve + // rejects; and an external sky-motion pass would decide the reactive + // policy for every cloud pixel in the frame. + HasExternalShader(report, "decals.vsh") || HasExternalShader(report, "decals.fsh") || + // Every stage Optimum owns outright: the resolve itself, the debug + // views, the sky-motion pass and the post-resolve sharpen. An + // external copy of any of them is not a writer that emits nothing, + // it is a replacement resolve running against a contract (MRT + // layout, history formats, jitter and reactive semantics) it cannot + // know. The prefix rule covers taa-* files added after this line was + // written, so a new stage cannot ship without a scanner rule. + HasExternalShader(report, "taa-resolve.vsh") || HasExternalShader(report, "taa-resolve.fsh") || + HasExternalShader(report, "taa-debug.vsh") || HasExternalShader(report, "taa-debug.fsh") || + HasExternalShader(report, "taa-skymotion.vsh") || HasExternalShader(report, "taa-skymotion.fsh") || + HasExternalShader(report, "taa-sharpen.vsh") || HasExternalShader(report, "taa-sharpen.fsh") || + HasExternalShaderPrefix(report, "taa-") || + // The sharpen pass is an RCAS variant and shares its vertex stage and + // lobe maths with the FSR1 pair, which is also what render scale + // resolves through: an external copy leaves TAA sharpening either + // doubled with FSR's own tap or gone. + HasExternalShader(report, "fsr-rcas.vsh") || HasExternalShader(report, "fsr-rcas.fsh") || + HasExternalShader(report, "fsr-easu.vsh") || HasExternalShader(report, "fsr-easu.fsh") || + // Not just vertexwarp.vsh: ShaderRegistry merges every shaderinclude + // into one dictionary that all the motion writers compile against, so + // an external file anywhere in that directory can redefine a helper + // the writers call - and unlike a shader, an include has no program + // of its own to point the blame at. + HasExternalShader(report, "vertexwarp.vsh") || + HasExternalShaderInclude(report); + AddFeatureDecision(report, "Taa", externalMotionShader, + "external shader owns a motion-vector writer contract"); + if (report.ScanFailed) { foreach (string feature in ShaderFeatures) AddFeatureDecision(report, feature, true, "scanner failure: conservative fallback"); + // A scan that did not finish cannot say which programs a mod replaced, + // so none of them may link the native SPIR-V over a mod's source. + report.RewriterPrograms = [AllPrograms]; } report.Fingerprint = ComputeFingerprint(report); @@ -336,6 +546,65 @@ private static bool HasExternalShader(ShaderCompatibilityReport report, string f string.Equals(Path.GetFileName(path), fileName, StringComparison.OrdinalIgnoreCase)); } + /// + /// True when any external shader file name starts with . + /// Optimum's own stages share the "taa-" prefix, so a stage added later is + /// covered without touching the feature decision. + /// + private static bool HasExternalShaderPrefix(ShaderCompatibilityReport report, string prefix) + { + return report.ShaderOwners.Keys.Any(path => + Path.GetFileName(path).StartsWith(prefix, StringComparison.OrdinalIgnoreCase)); + } + + /// + /// True when any external file lands in the shaderincludes directory. + /// NormalizeShaderPath keeps those under a "shaderincludes/" prefix. + /// + private static bool HasExternalShaderInclude(ShaderCompatibilityReport report) + { + return report.ShaderOwners.Keys.Any(path => + path.Replace('\\', '/').Contains("shaderincludes/", StringComparison.OrdinalIgnoreCase)); + } + + /// + /// One finding per external shader asset that replaces a known program's stage + /// by file name, or any shaderincludes file. The union lands in + /// : sorted program base + /// names, or the single entry when an include was + /// overridden, because ShaderRegistry merges every include into the one dictionary + /// all programs compile against and an include has no program of its own. + /// A mod shader with a new name is not an override: it has no native blob and + /// takes the rewriter anyway. + /// + private static void AddShaderAssetOverrides(ShaderCompatibilityReport report, HashSet? installedPrograms) + { + var known = installedPrograms ?? new HashSet(BuiltInPrograms, StringComparer.OrdinalIgnoreCase); + var programs = new SortedSet(StringComparer.OrdinalIgnoreCase); + bool all = false; + report.ShaderAssetOverrides.Clear(); + foreach ((string asset, List owners) in report.ShaderOwners.OrderBy(x => x.Key, StringComparer.OrdinalIgnoreCase)) + { + string normalized = asset.Replace('\\', '/'); + ShaderAssetOverride finding; + if (normalized.Contains("shaderincludes/", StringComparison.OrdinalIgnoreCase)) + { + all = true; + finding = new ShaderAssetOverride { Asset = asset, Programs = [AllPrograms] }; + } + else + { + string program = Path.GetFileNameWithoutExtension(normalized).ToLowerInvariant(); + if (!known.Contains(program)) continue; + programs.Add(program); + finding = new ShaderAssetOverride { Asset = asset, Programs = [program] }; + } + finding.Owners = [.. owners]; + report.ShaderAssetOverrides.Add(finding); + } + report.RewriterPrograms = all ? [AllPrograms] : [.. programs]; + } + private static bool IsShaderHookIndicator(string indicator) => indicator is "ShaderRegistry" or "LoadShader" or "ShaderProgram" or "Harmony" or "RenderHook"; @@ -375,6 +644,7 @@ private static bool IsOptimumSource(string path, string gameDir) string normalized = path.Replace('\\', '/'); int marker = normalized.IndexOf("assets/game/shaders/", StringComparison.OrdinalIgnoreCase); string shader; + bool isInclude = false; if (marker >= 0) { shader = normalized[marker..]; @@ -392,12 +662,45 @@ private static bool IsOptimumSource(string path, string gameDir) } else { - return null; + // Optimum: shaderincludes is a first-class asset category that + // ShaderRegistry merges into the same include dictionary as + // shaders, so an external vertexwarp.vsh replaces Optimum's copy + // exactly the way an external chunkopaque.vsh would - and with it + // the WarpState overloads the motion-vector writers evaluate. + marker = normalized.IndexOf("/shaderincludes/", StringComparison.OrdinalIgnoreCase); + if (marker >= 0) + { + shader = normalized[(marker + 1)..]; + isInclude = true; + } + else if (normalized.StartsWith("shaderincludes/", StringComparison.OrdinalIgnoreCase)) + { + shader = normalized; + isInclude = true; + } + else + { + return null; + } } } + // Optimum: the stage-extension filter is only meaningful for shaders/, + // where a file is a vertex, fragment or geometry stage. ShaderRegistry + // loads every shaderinclude regardless of extension - vanilla ships five + // .ash includes next to the .fsh/.vsh ones - so an external .ash override + // replaces a helper the motion writers compile against just the same. string extension = Path.GetExtension(shader); - if (extension is not ".fsh" and not ".vsh" and not ".gsh") return null; + if (isInclude) + { + // Any real file counts; a directory entry (no extension) does not. + if (extension.Length == 0) return null; + } + else if (extension is not ".fsh" and not ".vsh" and not ".gsh") + { + return null; + } + return shader.ToLowerInvariant(); } @@ -488,6 +791,35 @@ public sealed class ShaderCompatibilityReport public List DisabledFeatures { get; set; } = []; public Dictionary> FeatureReasons { get; set; } = new(StringComparer.OrdinalIgnoreCase); public List ScanErrors { get; set; } = []; + + /// External assets that replace a known program's stage or a shaderinclude. + public List ShaderAssetOverrides { get; set; } = []; + + /// + /// What the Vulkan runtime consumes: program base names that must be built through + /// the rewriter from the (mod) GLSL instead of the native SPIR-V, sorted; or the + /// single entry . Empty when + /// every program may link natively. + /// + public List RewriterPrograms { get; set; } = []; + + /// Sources that route the session to OpenGL (RawOpenGL or PlatformInternals). + public List OpenGlRequiredBy { get; set; } = []; + + /// True when the scan explicitly vetoes the Vulkan backend. + public bool OpenGlRequired => DisabledFeatures.Contains("Vulkan", StringComparer.OrdinalIgnoreCase); +} + +/// +/// ShaderAssetOverride finding: (normalized path, e.g. +/// assets/mymod/shaders/chunkopaque.fsh) replaces the listed programs' +/// source; is ["all"] for a shaderincludes file. +/// +public sealed class ShaderAssetOverride +{ + public string Asset { get; set; } = string.Empty; + public List Programs { get; set; } = []; + public List Owners { get; set; } = []; } public sealed class ShaderModSource diff --git a/Optimum.Patcher/ILPatcher.cs b/Optimum.Patcher/ILPatcher.cs index bd98c569..5f797ff9 100644 --- a/Optimum.Patcher/ILPatcher.cs +++ b/Optimum.Patcher/ILPatcher.cs @@ -67,7 +67,9 @@ public static int PatchWithInjection( List? hooks = null, Dictionary>? interfacesToInject = null, bool requireAllTargets = true, - Dictionary>? fieldsToRetype = null) + Dictionary>? fieldsToRetype = null, + List? typesToUnseal = null, + List? methodsToVirtualize = null) { var resolver = new DefaultAssemblyResolver(); resolver.AddSearchDirectory(Path.GetDirectoryName(vanillaPath)!); @@ -237,6 +239,18 @@ public static int PatchWithInjection( } } + // Phase 4: platform substitution attribute surgery. Runs after every body + // transplant and hook: TransplantBody never touches MethodAttributes today, but + // applying the flags last means no earlier phase can drop them. + int unsealedTypes = 0; + int virtualizedMethods = 0; + if (typesToUnseal != null && typesToUnseal.Count > 0) + unsealedTypes = PlatformSubstitution.UnsealTypes(vanillaAsm.MainModule, typesToUnseal); + if (methodsToVirtualize != null && methodsToVirtualize.Count > 0) + virtualizedMethods = PlatformSubstitution.VirtualizeMethods(vanillaAsm.MainModule, methodsToVirtualize).Count; + if (unsealedTypes + virtualizedMethods > 0) + Console.WriteLine($" Platform substitution: {unsealedTypes} types unsealed, {virtualizedMethods} methods virtualized."); + int requiredTargetCount = targets.Count(target => !target.Optional); Console.WriteLine( $"\n Summary: {injectedTypes} types, {injectedMembers} members, " + @@ -284,6 +298,23 @@ public static int PatchWithInjection( return -1; } + if (unsealedTypes + virtualizedMethods > 0) + { + var dispatchErrors = PlatformSubstitution.VerifyVirtualDispatch( + vanillaAsm.MainModule, + typesToUnseal ?? new List(), + methodsToVirtualize ?? new List(), + out int virtualCallSites); + if (dispatchErrors.Count > 0) + { + Console.Error.WriteLine($"\n {dispatchErrors.Count} virtual dispatch error(s), output not written:"); + foreach (var err in dispatchErrors) + Console.Error.WriteLine($" {err}"); + return -1; + } + Console.WriteLine($" Virtual dispatch verifier: ok, {virtualCallSites} callvirt/ldvirtftn sites reach virtualized methods, 0 call/ldftn."); + } + AssemblyWriter.Write(vanillaAsm, outputPath, preserveSymbols); if (preserveSymbols) { diff --git a/Optimum.Patcher/Program.cs b/Optimum.Patcher/Program.cs index a126f9e0..b0a0d550 100644 --- a/Optimum.Patcher/Program.cs +++ b/Optimum.Patcher/Program.cs @@ -54,6 +54,7 @@ "Optimum.EntityLightBatchBuffer", "Optimum.OptimumOptiTimeGuard", "Vintagestory.Client.NoObf.OptimumGreedyMeshEmitter", + "Vintagestory.Client.NoObf.OptimumAoClass", // Server-side worldgen scheduler + chunk read pool (see // docs/implementation-plans/server-worldgen-chunk-pool-cecil-wiring-plan-2026-08-11.md): // parallel SQLite read pool used by ChunkServerThread/ServerSystemSupplyChunks. @@ -63,6 +64,159 @@ // --- Phase 2b: Members to inject into existing types --- var membersToInject = new Dictionary> { + // Vulkan-native plan, Phase 1A: graphics bring-up virtuals VulkanClientPlatform overrides + // (injected with their flags, so they arrive virtual), and the ClientProgram.Start helpers + // that wire a platform and let the OpenGL fallback rebuild one. + ["Vintagestory.Client.NoObf.ClientPlatformAbstract"] = new() + { + "InitializeGraphics", + "ShutdownGraphics", + // Phase 1A step 2: the TAA/FSR members the renderers call without a cast to + // ClientPlatformWindows. Neutral bodies; ClientPlatformWindows overrides them. + "MotionAttachmentIndex", + "OptimumMotionWriteActive", + "TaaTargetsReady", + "TaaResolvedThisFrame", + "TaaHistory", + "BeginMotionWrite", + "EndMotionWrite", + "BeginMotionOnlyWrite", + "EndMotionOnlyWrite", + "RenderOptimumSkyMotion", + // Phase 3b stage 2: the sky dome's draw seam, so a native platform records that pass + // itself. The neutral body is the RenderMesh call it replaced. + "RenderSkyDome", + // Phase 3b stage 2: the chunk draw-group scope, so a native platform can state a + // terrain pipeline's fixed state instead of reading it back off the GL state. Neutral + // bodies: BeginChunkPass returns false and EndChunkPass does nothing, so the OpenGL + // path draws exactly what it drew before. + "BeginChunkPass", + "EndChunkPass", + // Phase 3b stage 2: the entity draw seam (every sub-mesh of a multi-texture mesh), so a + // native platform records the entityanimated and shadowmapentityanimated passes itself. + // The neutral body is the RenderMesh call it replaced. + "RenderEntityMesh", + // Phase 3b stage 2: the draw seams of the remaining sky, particle and decal systems, + // each with the neutral body of the draw it replaced. + "RenderNightSkyBox", + "RenderCelestialQuad", + "RenderSunQuad", + "RenderGuiQuad", + "RenderParticles", + // The decal pool draws through a scope seam, not a draw seam: the mesh handle stays + // inside MeshDataPool (internal in the vanilla API), so the lib runs the vanilla + // MeshDataPool.Draw between Begin and End and a native platform takes its RenderMesh + // multi-draw while the scope is open. Neutral bodies are empty. + "BeginDecalPass", + "EndDecalPass", + // Phase 3b stage 2, GUI and text: the two GUI draw seams, so a native platform records + // those passes itself. Both neutral bodies are the RenderMesh call they replaced. + "RenderTextureQuad", + "RenderOverlayLines", + "RenderOptimumTaaResolve", + "RenderOptimumTaaSharpen", + // Phase 3b stage 1d: the draw seams of the two TAA passes, so a native platform + // replaces the draw while the temporal contract stays in the lib body. + "OptimumTaaResolveDraw", + "OptimumTaaSharpenDraw", + "OptimumFsrBlitActive", + "DisableOptimumTaa", + // Optimum AO: the platform's own ambient occlusion (0 = vanilla SSAO) and its debug outputs. + "RenderOptimumAmbientOcclusion", + "OptimumAmbientOcclusionDebugTexture", + // Headless render harness: the channel order ReadDefaultFramebuffer leaves + // in the caller's buffer. B G R A on both backends: the OpenGL path reads + // GL_BGRA and VulkanClientPlatform converts its R8G8B8A8 texels to match. + "OptimumDefaultFramebufferIsBgra", + // Phase 1A step 3: the program, uniform and UBO operations ShaderProgramBase and + // UBO call. Neutral bodies; ClientPlatformWindows overrides them. SetUniform and + // SetUniformMatrix inject every overload the donor declares. + "UseShaderProgram", + "DisposeShaderProgram", + "BindSampler", + "SetUniform", + "SetUniformArray1", + "SetUniformArray2", + "SetUniformArray3", + "SetUniformArray4", + "SetUniformMatrix", + "SetUniformMatrices", + "SetUniformMatrices4x3", + "BindProgramTexture2D", + "BindProgramTextureCube", + "BindUBO", + "UnbindUBO", + "UpdateUBO", + "DeleteUBO", + // Phase 1A step 4: the frame bracket, window-size notification, thick-line probe + // and the graphics-API fragments of the framebuffer, post-chain and TAA methods. + // Neutral bodies; ClientPlatformWindows overrides them with the GL lines and + // VulkanClientPlatform with the device calls. + "BeginFrame", + "EndFrame", + "ProbeThickLineSupport", + "OnWindowSizeChanged", + "BindCurrentFrameBuffer", + "BindCurrentFrameBufferKeepViewport", + "ClearBoundFrameBuffer", + "ClearFrameBufferPass", + "ApplyTransparentPassBlendState", + "SelectBackDrawBuffer", + "SetBlendEnabled", + "ApplyTransparentMergeBlendState", + "ClearSsaoTarget", + "BeginFinalCompositionDrawBuffers", + "RestoreWorldDrawBuffers", + "EnableMotionDrawBuffers", + "RestorePrimaryDrawBuffers", + "EnableMotionOnlyDrawBuffers", + "ApplyOptimumMotionBlendState", + "ApplyOptimumMotionAccumulateBlendState", + "SelectFsrDrawBuffer", + "ReadTextureForParity", + // Phase 1A step 5: the leaf operations the render systems outside the platform + // issued (ScreenManager, ClientMain, VAO, ChunkRenderer, ShaderRegistry, the debug + // overlay, SvgLoader, InventoryItemRenderer, the OIT layers, the sun occlusion + // probe, Screenshot, ClientSystemStartup). Neutral bodies; ClientPlatformWindows + // overrides them with the GL lines and VulkanClientPlatform with device calls. + "SetDepthRange", + "ClearDefaultDepth", + "DeleteMeshHandle", + "DeleteVertexArrayHandles", + "SetTextureLodBias", + "SetSamplerLodBias", + "SetTextureDepthCompare", + "ClearTextureRegion", + "LoadTextureFromRgbaPointer", + "SetProgramSamplerUnit", + "CreateOitTargets", + "BeginOitAccumulation", + "BindOitTextures", + "GenOcclusionQuery", + "BeginOcclusionQuery", + "EndOcclusionQuery", + "TryGetOcclusionQueryResult", + "DeleteOcclusionQuery", + "ReadDefaultFramebuffer", + "GraphicsBackendName", + // Phase 2 (contract C3): the render-stage bracket ClientMain.TriggerRenderStage calls. + "BeginRenderStage", + "EndRenderStage", + // "Latency seams" S3: the pre-input sleep, the frame-cap ownership flag and the + // effective frame cap (background reduction included) window_RenderFrame calls. + "LatencySleep", + "LatencyOwnsFrameCap", + "SetLatencyFrameCap", + // World/UI separation: the compose ClientMain.RenderToDefaultFramebuffer and + // ScreenManager.Render call; neutral here, the Vulkan platform overrides it. + "OptimumComposeUiTarget", + }, + ["Vintagestory.Client.ClientProgram"] = new() + { + "ConfigureClientPlatform", + "WireClientPlatform", + "OptimumStartSinglePlayerServer", + }, ["Vintagestory.Client.NoObf.ClientSettings"] = new() { "OptimumEntityShadowCull", @@ -81,6 +235,16 @@ "OptimumDynamicLightCache", "OptimumRenderScale", }, + // TAA P4: the cube-particle motion writer's previous-frame uniforms. + ["Vintagestory.Client.NoObf.SystemRenderParticles"] = new() + { + "SetOptimumMotionUniforms", + }, + // TAA P4: the decal motion writer's previous-frame uniforms. + ["Vintagestory.Client.NoObf.SystemRenderDecals"] = new() + { + "SetOptimumMotionUniforms", + }, ["Vintagestory.Client.NoObf.SystemRenderPlayerEffects"] = new() { "GetOptimumLightRadius", @@ -100,6 +264,9 @@ "PrepareOptimumEntityLights", "BeginOptimumEntityShaderSegment", "EndOptimumEntityShaderSegment", + // TAA review fix: per-renderer motion-window gate. + "optimumMotionWriterTypes", + "OptimumIsMotionWriter", }, ["Vintagestory.Client.NoObf.ClientChunk"] = new() { @@ -121,6 +288,14 @@ }, ["Vintagestory.Client.NoObf.ClientPlatformWindows"] = new() { + "OptimumPostAmbientOcclusionTexture", + "OptimumPostSsaaLevel", + "OptimumPostSsaoInScene", + "OptimumTaaResolveDraw", + "OptimumTaaSharpenDraw", + "_optimumSharedIndirectCommands", + "_optimumSingleIndirectBufferCapacity", + "_optimumSingleIndirectBufferId", "_optimumSettingsInitialized", "EnsureOptimumDefaults", "_optimumTimerResolutionRaised", @@ -131,23 +306,227 @@ "_optimumFocusLostStopwatch", "optimumFsrDisabled", "DisableOptimumFsr", - "_optimumSingleIndirectBufferId", - "_optimumSingleIndirectBufferCapacity", - "_optimumSharedIndirectCommands", + // Phase 3b: the window's client size as a seam, so the native blit and the OpenGL body + // read the same value (docs/vulkan.md, decision 3). + "OptimumWindowClientSize", + // Phase 3b: the post chain split into one virtual per pass, so a native chain + // (VulkanClientPlatform.NativePostChain) owns the order and replaces one step at a + // time, plus the keep-the-viewport bind a native pass restores the GL-shaped state with. + "OptimumPostAmbientOcclusion", + "OptimumPostSceneTexture", + "OptimumPostGlowTexture", + "OptimumPostBloom", + "OptimumPostGodRays", + "OptimumPostLuma", + "OptimumPostFinish", + "OptimumBindKeepViewport", + // Phase 3b stage 1e-1g: the per-frame switches, the render scale and the two AO + // fields a native bloom, god-rays, Luma and final-composition pass reads as client + // state instead of GL state (decision 3). + "OptimumRenderBloom", + "OptimumRenderGodRays", + "OptimumRenderFxaa", + "OptimumSsaaLevel", + "OptimumAmbientOcclusionTexture", + "OptimumSsaoInScene", + // TAA: motion attachment, history/aux/prev-depth targets, and the + // debug-view blit path (P1). + // Phase 1A step 4: read by VulkanClientPlatform (GlToggleBlend, the Primary clear). + "OptimumRenderSsao", + "OptimumAdoptFrameBufferSettings", + "OptimumTaaRequested", + "OptimumSsaoKernel", + "SetOptimumMotionAttachmentIndex", + "OptimumAdoptTaaTargets", + "OptimumFinishDeviceFrameBufferSetup", + // Phase 1A step 4: GL halves of the framebuffer binding, clears and post-chain pass state. + "BindCurrentFrameBuffer", + "BindCurrentFrameBufferKeepViewport", + "ClearBoundFrameBuffer", + "ClearFrameBufferPass", + "ApplyTransparentPassBlendState", + "SelectBackDrawBuffer", + "SetBlendEnabled", + "ApplyTransparentMergeBlendState", + "ClearSsaoTarget", + "BeginFinalCompositionDrawBuffers", + "RestoreWorldDrawBuffers", + // Phase 1A step 4: the GL frame end, thick-line probe and parity readback. + "EndFrame", + "ProbeThickLineSupport", + "ReadTextureForParity", + // Phase 1A step 5: the GL halves of the leaf operations. + "SetDepthRange", + "ClearDefaultDepth", + "DeleteMeshHandle", + "DeleteVertexArrayHandles", + "SetTextureLodBias", + "SetSamplerLodBias", + "SetTextureDepthCompare", + "ClearTextureRegion", + "LoadTextureFromRgbaPointer", + "SetProgramSamplerUnit", + "CreateOitTargets", + "BeginOitAccumulation", + "BindOitTextures", + "GenOcclusionQuery", + "BeginOcclusionQuery", + "EndOcclusionQuery", + "TryGetOcclusionQueryResult", + "DeleteOcclusionQuery", + "ReadDefaultFramebuffer", + "GraphicsBackendName", + "OptimumTaaHistoryIndexA", + "OptimumTaaHistoryIndexB", + "OptimumGlR32f", + "MotionAttachmentIndex", + "TaaTargetsReady", + // Phase 1A step 2: the four state members above and below are overrides of + // ClientPlatformAbstract's virtuals now, reading these private fields. + "optimumMotionAttachmentIndex", + "optimumTaaTargetsReady", + "optimumTaaResolvedThisFrame", + "optimumMotionWriteActive", + "optimumTaaDisabled", + "TaaHistory", + "CreateOptimumHistoryTargetGl", + "DisableOptimumTaa", + "optimumTaaShaderReloadPending", + "OptimumRunPendingTaaShaderReload", + "_taaFrameParity", + "_taaHistoryValid", + "taaResolvedColorTexture", + "taaResolvedGlowTexture", + "TaaResolvedThisFrame", + "RenderOptimumTaaResolve", + // TAA P3: the motion-attachment draw-buffer window the terrain (and + // later entity/standard/instanced) writers open around their draws. + "OptimumMotionWriteActive", + "BeginMotionWrite", + "EndMotionWrite", + "ApplyOptimumMotionBlendState", + "InstallOptimumMotionWriteHooks", + "optimumMotionDrawBuffersOn", + "optimumMotionDrawBuffersOff", + // TAA P4: the motion-only window the liquid velocity pass opens - the + // motion attachment alone, every other colour attachment masked out. + "BeginMotionOnlyWrite", + "EndMotionOnlyWrite", + "optimumMotionOnlyDrawBuffers", + // Phase 1A step 4: the GL halves of the motion windows and the FSR target + // selection, overrides of ClientPlatformAbstract's virtuals. + "EnableMotionDrawBuffers", + "RestorePrimaryDrawBuffers", + "EnableMotionOnlyDrawBuffers", + "SelectFsrDrawBuffer", + // TAA P4: additive blending on the motion attachment for the OIT merge, + // which contributes the transparent layer's coverage to the reactive + // channel without touching the vector or the writer depth under it. + "ApplyOptimumMotionAccumulateBlendState", + // TAA P4: the sky / volumetric-cloud motion and reactive pass and the + // reactive constant it stamps. + "RenderOptimumSkyMotion", + "OptimumCloudReactive", + // TAA P5: the post-resolve sharpen pass, its dedicated target slot and + // the shared "is FSR's RCAS going to run this frame" test the pass and + // BlitPrimaryToDefault both ask so the two never sharpen the same + // pixels twice. + "OptimumTaaSharpenIndex", + // World/UI separation: the snapshot and UI slots the parity dump names. + "OptimumSceneNoHudIndex", + "OptimumUiTargetIndex", + "OptimumFsrBlitActive", + "RenderOptimumTaaSharpen", + // TAA: the jittered AO multiplied into the scene before the resolve, and + // the flag Final reads so the AO is never applied twice. + "optimumSsaoInScene", + "ApplyOptimumSceneSsao", + // Optimum AO: this frame's GTAO visibility texture (0 = vanilla SSAO), composed through + // ApplyOptimumSceneSsao, and the slots of its opt-in debug outputs in the parity dump. + "optimumAmbientOcclusionTexture", + "OptimumAoWorkingSlot", + "OptimumAoEdgesSlot", + "OptimumAoDepthSlot", + "OptimumAoOutputSlot", + "OptimumAoOutputCount", + // Phase 0 parity: the per-attachment dump (OPTIMUM_PARITY_DUMP) called from + // window_RenderFrame, its in-world frame counter, slot names, the single + // device-readback call site and the glGetTexImage body. + "optimumParityWorldFrames", + "optimumParityDumpDone", + "OptimumRunParityDump", + "OptimumParitySlotName", + "OptimumParityDumpAttachment", + "OptimumParityReadTextureGl", + // Phase 1A step 3: overrides of ClientPlatformAbstract's program, uniform and + // UBO virtuals, holding the device branch and GL lines ShaderProgramBase and UBO + // used to call directly. Every SetUniform/SetUniformMatrix overload is injected. + "UseShaderProgram", + "DisposeShaderProgram", + "BindSampler", + "SetUniform", + "SetUniformArray1", + "SetUniformArray2", + "SetUniformArray3", + "SetUniformArray4", + "SetUniformMatrix", + "SetUniformMatrices", + "SetUniformMatrices4x3", + "BindProgramTexture2D", + "BindProgramTextureCube", + "BindUBO", + "UnbindUBO", + "UpdateUBO", + "DeleteUBO", + }, + // TAA P3: the uniform block a buffer feeds and the point it is bound to. + // Vanilla had one block per program and Bind() hard-coded binding point 0; + // the entity motion writer adds a second ("AnimationPrev") beside it, and + // Update routes the bone upload through OptimumEntityMotion by block name. + ["Vintagestory.Client.NoObf.UBO"] = new() + { + "BlockName", + "BindingPoint", }, ["Vintagestory.Client.NoObf.ShaderPrograms"] = new() { "FsrEasu", "FsrRcas", + "TaaDebug", + "TaaResolve", + // TAA P5: the post-resolve sharpen pass program. + "TaaSharpen", + // TAA: the AO multiply into the scene before the resolve. + "SceneSsao", + // TAA P4: the liquid velocity pass program. + "ChunkLiquidMotion", + // TAA P4: the sky / volumetric-cloud motion pass program. + "TaaSkyMotion", + // World/UI separation: the UI compose pass program. + "UiCompose", }, ["Vintagestory.Client.NoObf.ShaderRegistry"] = new() { "RegisterOptimumShaderProgram", + // TAA: shared per-program post-compile handling extracted out of + // loadRegisteredShaderPrograms (both the parallel-preprocess and + // vanilla single-threaded paths call it); treats taa-debug as + // optional exactly like the two FSR programs. + "CompileAndTrackShaderProgram", + // TAA P5 review: the terrain sampler objects' LOD bias, reachable from + // ChunkRenderer so the TAA mip-bias row applies without a shader reload. + // A bound sampler object overrides the atlas texture parameter, so this + // is the only place chunkopaque/chunktopsoil mip selection changes. + "ApplyOptimumTerrainSamplerLodBias", + "ApplyOptimumSamplerLodBias", }, ["Vintagestory.Client.NoObf.SystemRenderOITLayers"] = new() { "optimumOitDisabled", "optimumOitFailureLogged", + // Phase 3b: the two OIT targets by handle, for the native OIT merge. + "OptimumOitRevealTexture", + "OptimumOitAccumTexture", "RestoreVanillaTransparentState", "DisableOptimumOit", }, @@ -155,6 +534,7 @@ ["Vintagestory.Client.NoObf.GuiCompositeSettings"] = new() { "oButtonBounds", + "vButtonBounds", "optimumContentBounds", "optimumRowY", "OnOptimumScrollChanged", @@ -163,6 +543,7 @@ "AddOptimumSliderRow", "AddOptimumDropdownRow", "OnOptimumOptions", + "OnVulkanOptions", "_AddOptimumTab", "onOptimumBackgroundFpsChanged", "onOptimumFramePacingChanged", @@ -179,10 +560,16 @@ "onOptimumChiselLodDistChanged", "onOptimumOcclusionScaleChanged", "onOptimumDynLightCacheChanged", + "onOptimumRendererChanged", "onOptimumEntityLightBatchChanged", "onOptimumEntityShaderCacheChanged", "onOptimumRenderScaleChanged", "onOptimumGodRaysCapChanged", + "onOptimumTaaChanged", + "onOptimumTaaSharpnessChanged", + "onOptimumTaaMipBiasChanged", + "onOptimumAmbientOcclusionChanged", + "onOptimumAmbientOcclusionDebugChanged", #if OPTIMUM_GREEDY_MESH "onOptimumGreedyMeshChanged", "onOptimumGreedySpanChanged", @@ -228,6 +615,12 @@ "edgePoolLocationsScratch", "optimumTextureLodBias", "ApplyOptimumTextureLodBias", + "SetOptimumTextureLodBias", + // TAA P3: previous-frame transforms for the terrain motion writers. + "SetOptimumMotionUniforms", + // TAA P4: the liquid velocity pass and its reactive constant. + "RenderLiquidMotion", + "OptimumLiquidReactive", }, // ChunkTesselatorManager: skip RecalcPriority+Sort when the player hasn't moved // (_lastSortPlayerPos/_lastSortYaw), plus the multi-tesselator worker pool and @@ -280,6 +673,24 @@ "RegisterTesselationThread", "GetTesselationWorkerSlot", "ChunkTesselatorManager", + // TAA P1: unjittered projection companion to CurrentProjectionMatrix + // (new property; the jittered getter itself is an existing transplant + // target below). + "CurrentProjectionMatrixUnjittered", + // TAA P5: OPTIMUM_FPS_LOG per-second frame-time line, read by + // scripts/dev/perf-capture.sh. Called from MainRenderLoop (a transplant + // target below); inert unless the env var names a file. + "optimumFpsLogPath", + "optimumFpsLogResolved", + "optimumFpsLogSamples", + "optimumFpsLogFrames", + "optimumFpsLogSeconds", + "OptimumLogFrameTime", + }, + ["Vintagestory.Client.NoObf.RenderAPIGame"] = new() + { + "CurrentProjectionMatrixUnjittered", + "TemporalContext", }, // Load-bearing dependency, wire before ServerSystemSupplyChunks: dispatchClaim's @@ -430,6 +841,9 @@ // --- Phase 1: Method bodies to transplant --- var targets = new List { + new("Vintagestory.Client.NoObf.ClientEventManager", "TriggerRenderStage", 2, new[] { "Vintagestory.API.Client.EnumRenderStage", "System.Single" }), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RenderMesh", 5, new[] { "Vintagestory.API.Client.MeshRef", "System.Int32[]", "System.Int32[]", "System.Int32", "System.Boolean" }), + new("Vintagestory.Client.NoObf.InventoryItemRenderer", "RenderItemstackToGui", 10, new[] { "Vintagestory.API.Common.ItemSlot", "System.Double", "System.Double", "System.Double", "System.Single", "System.Int32", "System.Single", "System.Boolean", "System.Boolean", "System.Boolean" }), // Entity render distance cull + shadow cull (reads injected ClientSettings.Optimum* props) new("Vintagestory.Client.NoObf.SystemRenderEntities", "OnRenderOpaque3D", 1), new("Vintagestory.Client.NoObf.SystemRenderEntities", "OnBeforeRender", 1), @@ -444,6 +858,35 @@ // FSR mip bias: refresh block atlas texture state after scale or atlas changes. new("Vintagestory.Client.NoObf.ChunkRenderer", "OnBeforeRenderOpaque", 1), new("Vintagestory.Client.NoObf.ChunkRenderer", "RuntimeAddBlockTextureAtlas", 1), + // TAA P3: terrain motion-vector writers - the opaque pass, the AfterOIT + // terrain overlay (pass 7) and the LiquidDepth prepass comment that records + // why it stays jittered but writes no motion. + new("Vintagestory.Client.NoObf.ChunkRenderer", "RenderAfterOIT", 1), + // Phase 3b stage 2 (native chunks): every ChunkRenderer draw group now brackets its + // pools with the BeginChunkPass/EndChunkPass seam, so the shadow cascades and the OIT + // groups are transplant targets too (RenderOpaque and RenderAfterOIT already are, and + // RenderLiquidMotion is an injected member). + new("Vintagestory.Client.NoObf.ChunkRenderer", "RenderShadow", 1), + new("Vintagestory.Client.NoObf.ChunkRenderer", "RenderOIT", 1), + new("Vintagestory.Client.NoObf.ChunkRenderer", "OnRenderBefore", 1), + // TAA P1: temporal frame contract - Advance()/JitterActive wiring in the + // render loop, the jittered projection getter, its capture at both + // Set3DProjection call sites, and the resets (FOV change, resize, world + // load already listed below as Start, shader reload). + new("Vintagestory.Client.NoObf.ClientMain", "MainRenderLoop", 1), + // Phase 2 (contract C3): brackets the stage's renderers with the platform's + // BeginRenderStage/EndRenderStage virtuals. + new("Vintagestory.Client.NoObf.ClientMain", "TriggerRenderStage", 2), + new("Vintagestory.Client.NoObf.ClientMain", "Set3DProjection", 2), + new("Vintagestory.Client.NoObf.ClientMain", "get_CurrentProjectionMatrix", 0), + new("Vintagestory.Client.NoObf.ClientMain", "OnFowChanged", 1), + new("Vintagestory.Client.NoObf.ClientMain", "OnResize", 0), + new("Vintagestory.Client.NoObf.ClientMain", "RenderAfterPostProcessing", 1), + // World/UI separation: the UI image is composed over the display image after the Ortho + // stage and before TriggerRenderStage(Done), where the with-HUD screenshot and the AVI + // writer are registered. + new("Vintagestory.Client.NoObf.ClientMain", "RenderToDefaultFramebuffer", 1), + new("Vintagestory.Client.NoObf.ClientEventManager", "TriggerReloadShaders", 0), // ClientMain: mouse wheel fix (vanilla fields only) new("Vintagestory.Client.NoObf.ClientMain", "OnMouseWheel", 1), // ClientMain: single-pass OpenedGuis scan instead of two LINQ calls (vanilla fields only) @@ -463,6 +906,15 @@ // the tesselation worker thread. See // docs/implementation-plans/chunk-tesselator-worker-pool-wiring-plan-2026-08-10.md. new("Vintagestory.Client.NoObf.ClientMain", "Start", 0), + // Phase 3b stage 2: Render2DTexture's GUI quads draw through RenderGuiQuad. Two overloads + // share a parameter count, so every target names its parameter types. + new("Vintagestory.Client.NoObf.ClientMain", "Render2DTexture", 8, + new[] { "Vintagestory.API.Client.MeshRef", "System.Int32", "System.Single", "System.Single", "System.Single", "System.Single", "System.Single", "Vintagestory.API.MathTools.Vec4f" }), + new("Vintagestory.Client.NoObf.ClientMain", "Render2DTexture", 7, + new[] { "Vintagestory.API.Client.MultiTextureMeshRef", "System.Single", "System.Single", "System.Single", "System.Single", "System.Single", "Vintagestory.API.MathTools.Vec4f" }), + new("Vintagestory.Client.NoObf.ClientMain", "Render2DTexture", 3, + new[] { "System.Int32", "Vintagestory.API.Common.ModelTransform", "Vintagestory.API.MathTools.Vec4f" }), + new("Vintagestory.Client.NoObf.ClientMain", "Render2DTextureFlipped", 7), // SystemRenderPlayerEffects: dynamic light radius (lambda-free rewrite) new("Vintagestory.Client.NoObf.SystemRenderPlayerEffects", "onBeforeRender", 1), // ClientPlatformWindows: persistent mapped VBO and index uploads. ParameterTypes @@ -482,22 +934,137 @@ new[] { "System.Int32[]", "System.Int32", "System.Int32", "Vintagestory.Client.NoObf.VAO", "System.Boolean" }), // ClientPlatformWindows: frame pacing + background FPS (inline in window_RenderFrame, no lambdas) new("Vintagestory.Client.NoObf.ClientPlatformWindows", "window_RenderFrame", 1), + // The shared index buffer is freed at shutdown, after the GL binding is gone + // on the device path; the raw call throws there instead of freeing it. + new("Vintagestory.Client.NoObf.ClientPlatformAbstract", "DisposeIndexBuffer", 0), // FSR: allocate the native intermediate and replace the final bilinear blit. new("Vintagestory.Client.NoObf.ClientPlatformWindows", "SetupDefaultFrameBuffers", 0), + // TAA P2: a framebuffer rebuild throws the history away - the flag and the + // temporal contract's reset reason are both set where the swap completes. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RebuildFrameBuffers", 0), new("Vintagestory.Client.NoObf.ClientPlatformWindows", "BlitPrimaryToDefault", 0), new("Vintagestory.Client.NoObf.ClientPlatformWindows", "DisableOptimumFsr", 1), // R4: pass the configured god-rays sample limit to the post-process shader. new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RenderPostprocessingEffects", 1), - // Issue #75 Tier 1: GPU indirect draw submission (glMultiDrawElementsIndirect) - new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RenderMesh", 5, - new[] { "Vintagestory.API.Client.MeshRef", "System.Int32[]", "System.Int32[]", "System.Int32", "System.Boolean" }), - // Issue #74: item-render profiler summary hook at the final render stage, and - // the reused-scratch per-slot item render path. - new("Vintagestory.Client.NoObf.ClientEventManager", "TriggerRenderStage", 2, - new[] { "Vintagestory.API.Client.EnumRenderStage", "System.Single" }), - // Issue #74 item-render profiler: measure per-slot GetItemStackRenderInfo cost. - new("Vintagestory.Client.NoObf.InventoryItemRenderer", "RenderItemstackToGui", 10, - new[] { "Vintagestory.API.Common.ItemSlot", "System.Double", "System.Double", "System.Double", "System.Single", "System.Int32", "System.Single", "System.Boolean", "System.Boolean", "System.Boolean" }), + // TAA P3: GlToggleBlend re-applies the motion attachment's replace blending. The other + // fixed-function bodies are vanilla again (Phase 1A step 4): VulkanClientPlatform + // overrides them, so they are no longer transplanted. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "GlToggleBlend", 2), + // Vulkan backend: the mod-facing uniform and texture-binding surface. A + // uniform location here is a byte offset into the generated block rather than + // a GL location, which callers never see. + // Uniform has seven two-parameter overloads, so each needs its signature. + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "System.Single" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "System.Int32" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "Vintagestory.API.MathTools.Vec2f" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "Vintagestory.API.MathTools.Vec2i" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "Vintagestory.API.MathTools.Vec3f" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "Vintagestory.API.MathTools.Vec3i" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 2, + new[] { "System.String", "Vintagestory.API.MathTools.Vec4f" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 3, + new[] { "System.String", "System.Int32", "System.Single[]" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 3, + new[] { "System.String", "System.Single", "System.Single" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 4, + new[] { "System.String", "System.Single", "System.Single", "System.Single" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniform", 5, + new[] { "System.String", "System.Single", "System.Single", "System.Single", "System.Single" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniforms2", 3), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniforms3", 3), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Uniforms4", 3), + // UniformMatrix has a float[] and a by-ref Matrix4 overload. + new("Vintagestory.Client.NoObf.ShaderProgramBase", "UniformMatrix", 2, + new[] { "System.String", "System.Single[]" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "UniformMatrix", 2, + new[] { "System.String", "OpenTK.Mathematics.Matrix4&" }), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "UniformMatrices", 3), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "UniformMatrices4x3", 3), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "BindTexture2D", 3), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "BindTextureCube", 3), + // Vulkan backend: program lifecycle. ProgramId is the device's handle. + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Use", 0), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Stop", 0), + new("Vintagestory.Client.NoObf.ShaderProgramBase", "Dispose", 0), + // Mesh update: the params-span-free CheckGlError format (Cecil constraint). The device + // halves of the mesh methods live in VulkanClientPlatform (Phase 1A step 4). + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "UpdateMesh", 2), + new("Vintagestory.Client.NoObf.VAO", "Dispose", 0), + // Vulkan backend: the window's own clear-and-swap has no GL binding to call + // when the window was opened with no graphics API. + new("Vintagestory.Client.NoObf.GameWindowNative", ".ctor", 2), + // Vulkan backend: framebuffer binding, lifecycle and per-pass state. The + // two CurrentFrameBuffer properties bind on assignment, so their setters + // are the seam for every render target the client selects. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "set_CurrentFrameBuffer", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "set_CurrentFrameBufferKeepVw", 1), + // The scissor flag is read back by the runtime atlas upload; the device + // keeps no queryable state, so the routed setter remembers it. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "DisposeFrameBuffers", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "ClearFrameBuffer", 4), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "ClearFrameBuffer", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "LoadFrameBuffer", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "UnloadFrameBuffer", 1, + new[] { "Vintagestory.API.Client.EnumFrameBuffer" }), + // Vulkan backend: startup capability reporting, which cannot ask GL. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "Start", 0), + // Optimum (headless render harness): a capture runs silent, so the mixer is + // created muted and every later attempt to restore the volume is answered with + // silence. Both bodies are vanilla's apart from that one condition. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "StartAudio", 0), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "set_MasterSoundLevel", 1), + // Vulkan backend: uniform buffers, whose handles UBO carries across. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "CreateUBO", 4), + new("Vintagestory.Client.NoObf.UBO", "Bind", 0), + new("Vintagestory.Client.NoObf.UBO", "Unbind", 0), + new("Vintagestory.Client.NoObf.UBO", "Dispose", 0), + new("Vintagestory.Client.NoObf.UBO", "Update", 3, + new[] { "System.Object", "System.Int32", "System.Int32" }), + // TAA P3: the second "AnimationPrev" uniform block for the skinned-entity + // motion writer, created beside "Animation" while TAA is on. + new("Vintagestory.Client.NoObf.ShaderProgramEntityanimated", "initUbos", 0), + // Vulkan backend: the packed-face storage buffer the SSBO chunk path uses. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "UpdateSSBOMesh", 2), + // Vulkan backend, world rendering: the render systems that reach past + // ClientPlatformWindows to GL directly. Each keeps its GL body behind a + // device check, the same shape as the platform routing. + new("Vintagestory.Client.NoObf.SystemRenderOITLayers/BeforeOIT", "rebuild", 0), + new("Vintagestory.Client.NoObf.SystemRenderOITLayers/BeforeOIT", "freeResources", 0), + new("Vintagestory.Client.NoObf.SystemRenderSunMoon", ".ctor", 1), + new("Vintagestory.Client.NoObf.SystemRenderSunMoon", "OnRenderFrame3DPost", 1), + new("Vintagestory.Client.NoObf.SystemRenderSunMoon", "Dispose", 1), + new("Vintagestory.Client.NoObf.SystemRenderFrameBufferDebug", "OnRenderFrame2DOverlay", 1), + new("Vintagestory.Client.NoObf.SvgLoader", "LoadSvg", 6), + new("Vintagestory.Client.NoObf.ClientMain", "OrthoMode", 3), + new("Vintagestory.Client.NoObf.ClientMain", "PerspectiveMode", 0), + new("Vintagestory.Client.NoObf.InventoryItemRenderer", "RenderItemStackToFrameBuffer", 3), + new("Vintagestory.Client.NoObf.ClientSystemStartup", "HandleLevelFinalize", 1), + new("Vintagestory.ClientNative.Screenshot", "GrabScreenshot", 4), + // Vulkan backend: the GUI depth clear between the world and the interface, + // the only raw GL left in the screen loop. + new("Vintagestory.Client.ScreenManager", "Render", 1), + // Vulkan backend: the post-process chain's remaining direct GL - viewport, + // draw-buffer selection, depth toggle and the SSAO clear. + // Vulkan backend: the device's default target and swapchain follow the + // window only if something tells them to. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "Window_Resize", 0), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "MergeTransparentRenderPass", 0), + // TAA P4: the cube-particle motion window and its uniforms. + new("Vintagestory.Client.NoObf.SystemRenderParticles", "OnRenderFrame3D", 1), + // Phase 3b stage 2: the pools' instanced draws go through the platform's particle seam. + new("Vintagestory.Client.NoObf.SystemRenderParticles", "Render", 2), + // Phase 3b stage 2: the star box and the moon draw through their own platform seams. + new("Vintagestory.Client.NoObf.SystemRenderNightSky", "OnRenderFrame3D", 1), + new("Vintagestory.Client.NoObf.SystemRenderSunMoon", "OnRenderFrame3D", 1), + // TAA P4: the decal motion window. + new("Vintagestory.Client.NoObf.SystemRenderDecals", "OnRenderFrame3D", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RenderFinalComposition", 0), // GuiCompositeMainMenuLeft: Optimum link in main menu (no lambdas) new("Vintagestory.Client.GuiCompositeMainMenuLeft", "Compose", 0), // E3: particle spawn distance gate, before the per-particle revive loop @@ -539,6 +1106,11 @@ new("Vintagestory.Client.NoObf.AmbientManager", "updateColorGradingValues", 1), // SystemRenderSkyColor: reusable scratch vectors instead of per-frame Vec3f allocations new("Vintagestory.Client.NoObf.SystemRenderSkyColor", "OnRenderFrame3D", 1), + // Phase 3b stage 2, GUI and text: the two callers that draw through the new GUI seams - + // the texture-into-texture blit that bakes every Cairo GUI and text surface, and the + // aiming reticle's line draws. + new("Vintagestory.Client.NoObf.ClientMain", "RenderTextureIntoFrameBuffer", 10), + new("Vintagestory.Client.NoObf.SystemRenderPlayerAimAcc", "OnRenderFrame2DOverlay", 1), // SystemSoundEngine: audio listener update threshold + periodic refresh new("Vintagestory.Client.NoObf.SystemSoundEngine", "OnRenderFrame", 2), // RenderAPIBase: skip disposed meshrefs instead of rendering freed GL handles (#8881/#8950/#8982-class crash) @@ -578,6 +1150,9 @@ new("Vintagestory.Client.NoObf.ChunkTesselator", "BuildBlockPolygons_EdgeOnly", 3), new("Vintagestory.Client.NoObf.ChunkTesselator", "BuildDecorPolygons", 5), new("Vintagestory.Client.NoObf.ChunkTesselator", "GetMeshPoolForPass", 3), + // GTAO thin class: stamp cross quads independently of snow and JSON faces. + new("Vintagestory.Client.NoObf.CrossTesselator", "DrawCross", 2), + new("Vintagestory.Client.NoObf.JsonTesselator", "AddJsonModelDataToMesh", 7), new("Vintagestory.Client.NoObf.TesselatedChunkPart", "AddModelAndStoreLocation", 8), // Eco Machina anchors its tapered-tree transpiler on this method's local slots. new("Vintagestory.Client.NoObf.ChunkTesselator", "CalculateVisibleFaces", 4), @@ -670,6 +1245,28 @@ new("Vintagestory.Server.ServerPackets", "GetBulkEntityDebugAttributesPacket", 1), }; +// --- Platform substitution (Vulkan-native plan, Phase 0) --- +// Optimum.Render.Vulkan ships VulkanClientPlatform : ClientPlatformWindows and overrides +// these members. ILPatcher applies both lists after every body transplant and hook, so a +// transplant of the same methods cannot drop the flags, then fails the patch if any body +// still reaches a virtualized method with `call`/`ldftn` (a caller that would bypass the +// override). Entries must be public or protected: a private virtual cannot be overridden +// from another assembly. +var typesToUnseal = new List +{ + "Vintagestory.Client.NoObf.ClientPlatformWindows", +}; + +var methodsToVirtualize = new List +{ + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "SetupDefaultFrameBuffers", 0), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "DisposeFrameBuffers", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "RenderFullscreenTriangle", 1), + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "GetGraphicsCardRenderer", 0), + // Phase 1A step 4: VulkanClientPlatform logs the device facts instead. + new("Vintagestory.Client.NoObf.ClientPlatformWindows", "LogAndTestHardwareInfosStage2", 0), +}; + int total = ILPatcher.PatchWithInjection( vanillaPath, compiledPath, outputPath, typesToInject, membersToInject, targets, @@ -707,7 +1304,9 @@ TargetGenericArity: 0, InsertBeforeTarget: true), }, - fieldsToRetype: fieldsToRetype); + fieldsToRetype: fieldsToRetype, + typesToUnseal: typesToUnseal, + methodsToVirtualize: methodsToVirtualize); Console.WriteLine($"\nDone."); return total > 0 ? 0 : 1; diff --git a/Optimum.Patcher/mod-patcher.cs b/Optimum.Patcher/mod-patcher.cs index 28cc8566..11ea6771 100644 --- a/Optimum.Patcher/mod-patcher.cs +++ b/Optimum.Patcher/mod-patcher.cs @@ -135,6 +135,15 @@ private static Manifest EssentialsManifest() }, Methods: [ + // Vulkan backend: the cloud renderers call OpenGL directly for + // their map framebuffer, tile textures and state; on the device + // path those go through the render seam instead. + new("FluffyClouds.CloudRendererMap", "FreeGlResources", 0), + new("FluffyClouds.CloudRendererMap", "OnRenderFrame", 2), + new("FluffyClouds.CloudRendererMap", "WriteTexture", 0), + new("FluffyClouds.CloudRendererMap", "makeTexture", 4), + new("FluffyClouds.CloudRendererMap", "InitCloudTiles", 1), + new("FluffyClouds.CloudRendererVolumetric", "OnRenderFrame", 2), new("Vintagestory.GameContent.BlockEntityParticleEmitter", "OnGameTick", 1), new("Vintagestory.GameContent.EntityBehaviorCollectEntities", "OnGameTick", 1), new("Vintagestory.GameContent.EntityBehaviorRepulseAgents", "OnGameTick", 1), @@ -148,6 +157,21 @@ private static Manifest EssentialsManifest() new("Vintagestory.GameContent.EntityShapeRenderer", ".ctor", 2), new("Vintagestory.GameContent.EntityShapeRenderer", "BeforeRender", 1), new("Vintagestory.GameContent.EntityShapeRenderer", "DoRender3DOpaqueBatched", 2), + // TAA P3: the first-person hands draw entity geometry outside the + // shared entity pass, with their own program, their own FOV and + // their own copy of the animation blocks, so both methods carry + // motion-writer changes. + new("Vintagestory.GameContent.EntityPlayerShapeRenderer", "DoRender3DOpaque", 2), + new("Vintagestory.GameContent.ModSystemFpHands", "LoadShaders", 0), + // TAA P3: the standard-shader motion writer. Held items (both hands, + // and the first-person item program) get their previous transform in + // RenderItem; dropped items in EntityItemRenderer.DoRender3DOpaque + // above. Both also open the motion-attachment window around their draw. + new("Vintagestory.GameContent.EntityShapeRenderer", "RenderItem", 5), + // TAA P4: the falling-block renderer is shared by every falling + // block in view, so each block keys its own previous model matrix + // on its entity and the window is opened once around the loop. + new("Vintagestory.GameContent.ModSystemRenderFallingBlocksFast", "OnRenderFrame", 2), new("Vintagestory.GameContent.WeatherSimulationParticles", "asyncParticleSpawn", 2), new("Vintagestory.GameContent.WeatherSystemClient", "OnRenderFrame", 2), new("Vintagestory.GameContent.WeatherSimulationSound", "updateSounds", 1), @@ -235,6 +259,53 @@ private static Manifest SurvivalManifest() new("Vintagestory.GameContent.GearRenderer", "LoadShader", 0), new("Vintagestory.GameContent.GearRenderer", "OnRenderFrame", 2), new("Vintagestory.GameContent.GearRenderer", "updateSuperMechState", 2), + // TAA P3: the echo chamber draws three meshes on the shared + // entityanimated program from DoRender3DOpaque, so it opens the + // motion-attachment window itself. + new("Vintagestory.GameContent.EchoChamberRenderer", "DoRender3DOpaque", 2), + // TAA P3: the quern top is the block-entity model that actually + // moves, so it keeps a previous model matrix and writes motion. + new("Vintagestory.GameContent.QuernTopRenderer", "OnRenderFrame", 2), + // TAA P4: the remaining moving standard-shader block-entity + // renderers. Each keeps a previous model matrix through + // OptimumStandardMotion and opens the motion-attachment window + // around its own draw; without these entries the installed runtime + // keeps the vanilla bodies and they ghost on the camera fallback. + new("Vintagestory.GameContent.HelveHammerRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.FruitpressContentsRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.ResonatorRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.BloomeryContentsRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.ForgeContentsRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.FirepitContentsRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.PotInFirepitRenderer", "OnRenderFrame", 2), + // TAA P3, the instanced writer: every mechanical-power renderer now + // fills OptimumInstanceMotion's instance layout (light, transform, + // previous transform, metadata) instead of vanilla's light+transform, + // so the buffer allocations, the transform writers and the instance + // counts all move together. MechNetworkRenderer sets the pass uniforms + // and opens the draw-buffer window around the whole loop. + new("Vintagestory.GameContent.Mechanics.MechNetworkRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.MechBlockRenderer", "UpdateCustomFloatBuffer", 0), + new("Vintagestory.GameContent.Mechanics.MechBlockRenderer", "UpdateLightAndTransformMatrix", 7), + new("Vintagestory.GameContent.Mechanics.GenericMechBlockRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.GenericMechBlockRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.AngledCageGearRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.AngledCageGearRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.AngledGearsBlockRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.AngledGearsBlockRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.TransmissionBlockRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.TransmissionBlockRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.ClutchBlockRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.ClutchBlockRenderer", "UpdateLightAndTransformMatrix", 9), + new("Vintagestory.GameContent.Mechanics.ClutchBlockRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.CreativeRotorRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.CreativeRotorRenderer", "createCustomFloats", 1), + new("Vintagestory.GameContent.Mechanics.CreativeRotorRenderer", "UpdateLightAndTransformMatrix", 8), + new("Vintagestory.GameContent.Mechanics.CreativeRotorRenderer", "OnRenderFrame", 2), + new("Vintagestory.GameContent.Mechanics.PulverizerRenderer", ".ctor", 4), + new("Vintagestory.GameContent.Mechanics.PulverizerRenderer", "createCustomFloats", 1), + new("Vintagestory.GameContent.Mechanics.PulverizerRenderer", "UpdateLightAndTransformMatrix", 8), + new("Vintagestory.GameContent.Mechanics.PulverizerRenderer", "OnRenderFrame", 2), ]); } diff --git a/Optimum.Patcher/platform-substitution.cs b/Optimum.Patcher/platform-substitution.cs new file mode 100644 index 00000000..af13a960 --- /dev/null +++ b/Optimum.Patcher/platform-substitution.cs @@ -0,0 +1,220 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Mono.Cecil; +using Mono.Cecil.Cil; + +namespace Optimum.Patcher; + +/// +/// Attribute surgery for platform substitution: a renderer assembly subclasses a vanilla +/// type (ClientPlatformWindows) and overrides members that vanilla declares sealed or +/// non-virtual. clears TypeAttributes.Sealed, +/// turns a non-virtual method into a new virtual slot, and +/// proves no method body still reaches a virtualized +/// method with a non-virtual call (or ldftn): such a caller would silently +/// bypass the override, which is the one failure mode the CLR never reports. +/// +public static class PlatformSubstitution +{ + private const MethodAttributes VirtualFlags = + MethodAttributes.Virtual | MethodAttributes.NewSlot | MethodAttributes.HideBySig; + + public static int UnsealTypes(ModuleDefinition module, IReadOnlyList typeNames) + { + int unsealed = 0; + foreach (string typeName in typeNames) + { + TypeDefinition type = module.GetType(typeName) + ?? throw new InvalidOperationException($"Type to unseal not found: {typeName}"); + if (type.IsInterface || type.IsValueType) + throw new InvalidOperationException($"Type to unseal is not a class: {typeName}"); + // A static class is abstract|sealed; unsealing it would not make it subclassable. + if (type.IsAbstract && type.IsSealed) + throw new InvalidOperationException($"Type to unseal is a static class: {typeName}"); + type.Attributes &= ~TypeAttributes.Sealed; + unsealed++; + } + return unsealed; + } + + public static List VirtualizeMethods(ModuleDefinition module, IReadOnlyList targets) + { + var virtualized = new List(); + foreach (MethodTarget target in targets) + { + MethodDefinition method = FindTarget(module, target); + string? visibility = OverridableVisibilityError(method); + if (visibility != null) + throw new InvalidOperationException( + $"Method to virtualize is {visibility} and cannot be overridden from another assembly: {target}"); + if (method.IsStatic || method.IsConstructor) + throw new InvalidOperationException($"Method to virtualize is static or a constructor: {target}"); + if (method.IsVirtual && (!method.IsNewSlot || method.IsFinal)) + throw new InvalidOperationException( + $"Method to virtualize already overrides a base slot or is final: {target}"); + + method.Attributes |= VirtualFlags; + virtualized.Add(method); + } + return virtualized; + } + + /// + /// Returns one error per violation: a type that still carries Sealed, a method without + /// Virtual, and every call/ldftn whose operand resolves to a virtualized + /// method. A call from a subclass of the declaring type (a base.X() call) + /// is legitimate and accepted. + /// + public static List VerifyVirtualDispatch( + ModuleDefinition module, + IReadOnlyList unsealedTypes, + IReadOnlyList virtualizedMethods, + out int virtualCallSites) + { + var errors = new List(); + virtualCallSites = 0; + + foreach (string typeName in unsealedTypes) + { + TypeDefinition? type = module.GetType(typeName); + if (type == null) + errors.Add($"unsealed type {typeName} is missing from the output module"); + else if (type.IsSealed) + errors.Add($"unsealed type {typeName} still carries TypeAttributes.Sealed"); + } + + var definitions = new List(); + foreach (MethodTarget target in virtualizedMethods) + { + MethodDefinition method; + try + { + method = FindTarget(module, target); + } + catch (InvalidOperationException error) + { + errors.Add(error.Message); + continue; + } + if (!method.IsVirtual) + errors.Add($"virtualized method {method.FullName} is not virtual"); + definitions.Add(method); + } + if (definitions.Count == 0) + return errors; + + var names = new HashSet(StringComparer.Ordinal); + foreach (MethodDefinition method in definitions) + names.Add(method.Name); + + var typesByName = new Dictionary(StringComparer.Ordinal); + var allTypes = new List(); + foreach (TypeDefinition type in module.Types) + Collect(type, allTypes, typesByName); + + foreach (TypeDefinition type in allTypes) + { + foreach (MethodDefinition caller in type.Methods) + { + if (!caller.HasBody) continue; + foreach (Instruction instruction in caller.Body.Instructions) + { + OpCode opCode = instruction.OpCode; + if (opCode.Code != Code.Call && opCode.Code != Code.Callvirt && + opCode.Code != Code.Ldftn && opCode.Code != Code.Ldvirtftn) + continue; + if (instruction.Operand is not MethodReference reference || !names.Contains(reference.Name)) + continue; + + MethodDefinition? resolved = Match(definitions, reference); + if (resolved == null) continue; + + if (opCode.Code == Code.Callvirt || opCode.Code == Code.Ldvirtftn) + { + virtualCallSites++; + continue; + } + if (opCode.Code == Code.Call && IsStrictSubclass(type, resolved.DeclaringType, typesByName)) + continue; + + errors.Add( + $"{caller.FullName} reaches virtualized {resolved.FullName} with {opCode.Name} " + + $"at IL_{instruction.Offset:X4}; an override would be bypassed"); + } + } + } + + return errors; + } + + private static MethodDefinition FindTarget(ModuleDefinition module, MethodTarget target) + { + TypeDefinition type = module.GetType(target.TypeFullName) + ?? throw new InvalidOperationException($"Type of method to virtualize not found: {target}"); + MethodDefinition? found = null; + foreach (MethodDefinition method in type.Methods) + { + if (method.Name != target.MethodName || method.Parameters.Count != target.ParamCount || !target.Matches(method)) + continue; + if (found != null) + throw new InvalidOperationException($"Ambiguous method to virtualize: {target}"); + found = method; + } + return found ?? throw new InvalidOperationException($"Method to virtualize not found: {target}"); + } + + private static string? OverridableVisibilityError(MethodDefinition method) + { + switch (method.Attributes & MethodAttributes.MemberAccessMask) + { + case MethodAttributes.Public: + case MethodAttributes.Family: + case MethodAttributes.FamORAssem: + return null; + case MethodAttributes.Private: + return "private"; + case MethodAttributes.Assembly: + return "internal"; + case MethodAttributes.FamANDAssem: + return "private protected"; + default: + return "compiler-controlled"; + } + } + + private static MethodDefinition? Match(List definitions, MethodReference reference) + { + foreach (MethodDefinition definition in definitions) + { + if (MethodSignature.Matches(definition, reference)) + return definition; + } + return null; + } + + private static bool IsStrictSubclass( + TypeDefinition type, TypeDefinition baseType, Dictionary typesByName) + { + TypeReference? current = type.BaseType; + int guard = 0; + while (current != null && guard++ < 64) + { + string name = current is GenericInstanceType generic ? generic.ElementType.FullName : current.FullName; + if (name == baseType.FullName) + return true; + if (!typesByName.TryGetValue(name, out TypeDefinition? next)) + return false; + current = next.BaseType; + } + return false; + } + + private static void Collect(TypeDefinition type, List all, Dictionary byName) + { + all.Add(type); + byName[type.FullName] = type; + foreach (TypeDefinition nested in type.NestedTypes) + Collect(nested, all, byName); + } +} diff --git a/Optimum.Render.Vulkan.Tests/AmbientOcclusion/SsaoTests.cs b/Optimum.Render.Vulkan.Tests/AmbientOcclusion/SsaoTests.cs new file mode 100644 index 00000000..0d2f35ec --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/AmbientOcclusion/SsaoTests.cs @@ -0,0 +1,609 @@ +// Source: Optimum.Render.Vulkan.Tests/SceneSsaoTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Runtime.InteropServices; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +/// +/// The AO multiply the TAA path runs before the resolve: scene-ssao draws into Primary +/// with only colour 0 selected and the Multiply blend, so the resolve accumulates the +/// occlusion together with the scene. Every other Primary attachment and the depth have +/// to come out untouched, frame after frame. +/// +public class SceneSsaoTests(ITestOutputHelper output) +{ + [SkippableTheory] + [InlineData(1)] + [InlineData(2)] + public unsafe void OcclusionMultipliesOnlySceneColourBeforeTheResolve(int quality) + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + const int size = 8, frames = 8; + var files = ShaderCorpus.LoadShaderFiles(); + int program = GpuTest.LinkProgram( + seam, + files["scene-ssao.vsh"], + files["scene-ssao.fsh"].Replace("#version 330 core", "#version 330 core\n#define SSAOLEVEL " + quality), + "scene-ssao" + quality); + + // Rows alternate between two AO levels, so SSAOLEVEL 2's min with the row + // above is visible: every row sees the darker of the two. + var ao = new byte[size * size * 4]; + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + for (int c = 0; c < 4; c++) ao[(y * size + x) * 4 + c] = (byte)(y % 2 == 0 ? 64 : 192); + int occlusion; + fixed (byte* data = ao) occlusion = seam.CreateTexture2DRaw(size, size, 0x8058, (IntPtr)data, 4); + + var colors = new int[frames][]; + var depths = new int[frames]; + var targets = new int[frames]; + for (int i = 0; i < frames; i++) + { + targets[i] = seam.CreateFramebuffer(size, size); + colors[i] = new int[5]; + for (int slot = 0; slot < 5; slot++) + { + colors[i][slot] = seam.CreateTexture2DRaw(size, size, 0x8058, IntPtr.Zero, 4); + seam.AttachTexture(targets[i], (EnumFramebufferAttachment)(36064 + slot), colors[i][slot], 0); + } + depths[i] = seam.CreateTexture2D(size, size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + seam.AttachTexture(targets[i], EnumFramebufferAttachment.DepthAttachment, depths[i], 0); + } + + seam.SetViewport(0, 0, size, size); + seam.SetCullFace(false); + seam.SetDepthTest(false); + seam.SetSamplerUnit(program, "ssaoScene", 0); + seam.SetUniform(program, seam.GetUniformLocation(program, "invRenderHeight"), 1f / size); + for (int i = 0; i < frames; i++) + { + seam.BeginFrame(); + seam.BindFramebuffer(targets[i]); + seam.SetDrawBuffers(targets[i], 31); + seam.ClearColor(0, (i + 1) / 16f, (i + 1) / 16f, (i + 1) / 16f, 1); + for (int slot = 1; slot < 5; slot++) seam.ClearColor(slot, slot / 8f, slot / 8f, slot / 8f, 1); + seam.ClearDepth(0.375f); + seam.SetDrawBuffers(targets[i], 1); + seam.SetBlend(true, EnumBlendMode.Multiply); + seam.UseProgram(program); + seam.BindTexture(0, occlusion); + seam.DrawFullscreenTriangle(); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetDrawBuffers(targets[i], 31); + seam.Present(); + } + + // The sequence completes before any readback or CPU wait. + seam.BeginFrame(); + for (int i = 0; i < frames; i++) + { + for (int slot = 0; slot < 5; slot++) + { + byte[] pixels = seam.ReadBackLevel0ForTests(colors[i][slot]); + for (int y = 1; y < size; y++) + for (int x = 0; x < size; x++) + { + double factor = quality == 2 || y % 2 == 0 ? 64 : 192; + double expected = slot == 0 ? Math.Round((i + 1) / 16.0 * 255) * factor / 255 : slot / 8.0 * 255; + int offset = (y * size + x) * 4; + for (int c = 0; c < 3; c++) Assert.InRange((double)pixels[offset + c], expected - 1.1, expected + 1.1); + Assert.Equal(255, pixels[offset + 3]); + } + } + foreach (float depth in MemoryMarshal.Cast(seam.ReadBackLevel0ForTests(depths[i]))) + Assert.Equal(0.375f, depth); + } + seam.Present(); + GpuTest.AssertClean(seam); + } + } + + /// + /// The OPTIMUMAO variant (docs/vulkan.md#ambient-occlusion C.9, C.11): the GTAO visibility is + /// fetched at render resolution with no min-of-two-rows, attenuated by vanilla's water, fog and + /// OIT term (gPosition.w + 0.75 * (1 - revealage)), and still multiplies only colour 0. + /// + [SkippableFact] + public unsafe void GtaoVisibilityIsComposedWithTheVanillaAttenuationAndNoRowMin() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + const int size = 8, frames = 4; + const float positionW = 0.25f, revealage = 0.6f; + var files = ShaderCorpus.LoadShaderFiles(); + int program = GpuTest.LinkProgram( + seam, + files["scene-ssao.vsh"], + files["scene-ssao.fsh"].Replace("#version 330 core", "#version 330 core\n#define SSAOLEVEL 2\n#define OPTIMUMAO 1"), + "scene-ssao-gtao"); + + var ao = new byte[size * size * 4]; + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + for (int c = 0; c < 4; c++) ao[(y * size + x) * 4 + c] = (byte)(y % 2 == 0 ? 64 : 192); + int visibility; + fixed (byte* data = ao) visibility = seam.CreateTexture2DRaw(size, size, 0x8058, (IntPtr)data, 4); + + int position = seam.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int reveal = seam.CreateTexture2DRaw(size, size, 0x8058, IntPtr.Zero, 4); + int inputs = seam.CreateFramebuffer(size, size); + seam.AttachTexture(inputs, EnumFramebufferAttachment.ColorAttachment0, position, 0); + seam.AttachTexture(inputs, EnumFramebufferAttachment.ColorAttachment1, reveal, 0); + seam.SetDrawBuffers(inputs, 3); + + var colors = new int[frames][]; + var targets = new int[frames]; + for (int i = 0; i < frames; i++) + { + targets[i] = seam.CreateFramebuffer(size, size); + colors[i] = new int[3]; + for (int slot = 0; slot < 3; slot++) + { + colors[i][slot] = seam.CreateTexture2DRaw(size, size, 0x8058, IntPtr.Zero, 4); + seam.AttachTexture(targets[i], (EnumFramebufferAttachment)(36064 + slot), colors[i][slot], 0); + } + } + + seam.SetViewport(0, 0, size, size); + seam.SetCullFace(false); + seam.SetDepthTest(false); + seam.SetSamplerUnit(program, "ssaoScene", 0); + seam.SetSamplerUnit(program, "gPositionScene", 1); + seam.SetSamplerUnit(program, "revealageScene", 2); + seam.SetUniform(program, seam.GetUniformLocation(program, "invRenderHeight"), 1f / size); + seam.SetUniform(program, seam.GetUniformLocation(program, "optimumAoMode"), 1); + for (int i = 0; i < frames; i++) + { + seam.BeginFrame(); + seam.BindFramebuffer(inputs); + seam.ClearColor(0, 0f, 0f, 0f, positionW); + seam.ClearColor(1, revealage, 0f, 0f, 1f); + seam.BindFramebuffer(targets[i]); + seam.SetDrawBuffers(targets[i], 7); + seam.ClearColor(0, 0.75f, 0.75f, 0.75f, 1f); + seam.ClearColor(1, 0.25f, 0.25f, 0.25f, 1f); // glow: never touched + seam.ClearColor(2, 0.5f, 0.5f, 0.5f, 1f); + seam.SetDrawBuffers(targets[i], 1); + seam.SetBlend(true, EnumBlendMode.Multiply); + seam.UseProgram(program); + seam.BindTexture(0, visibility); + seam.BindTexture(1, position); + seam.BindTexture(2, reveal); + seam.DrawFullscreenTriangle(); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetDrawBuffers(targets[i], 7); + seam.Present(); + } + + double attenuate = positionW + (1.0 - Math.Round(revealage * 255) / 255.0) * 0.75; + seam.BeginFrame(); + for (int i = 0; i < frames; i++) + { + for (int slot = 0; slot < 3; slot++) + { + byte[] pixels = seam.ReadBackLevel0ForTests(colors[i][slot]); + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + { + double v = (y % 2 == 0 ? 64 : 192) / 255.0; + double factor = Math.Clamp(1.0 - (1.0 - v) * (1.0 - attenuate), 0.0, 1.0); + double expected = slot switch + { + 0 => Math.Round(0.75 * 255) * factor, + 1 => Math.Round(0.25 * 255), + _ => Math.Round(0.5 * 255), + }; + int offset = (y * size + x) * 4; + for (int c = 0; c < 3; c++) Assert.InRange((double)pixels[offset + c], expected - 1.1, expected + 1.1); + } + } + } + seam.Present(); + GpuTest.AssertClean(seam); + } + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/SsaoNoiseAlphaTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan; +using Optimum.Render.Vulkan.Platform; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Xunit; +using Xunit.Abstractions; + +/// +/// Parity gap from the Phase 0 and Phase 1 dumps: framebuffer slot 13 (SSAO) colour +/// attachment 1, the 16x16 rotation-noise texture, read alpha 1.0 in every texel on +/// OpenGL and 0.0 on Vulkan. +/// +/// Not a masked write (no shader renders into that texture): the GL path allocates +/// GL_RGBA32F and uploads GL_RGB float data, and GL's pixel transfer fills the absent +/// alpha with 1. The device path uploaded four channels with a padding 0, and the device +/// copies channels verbatim. The fix is in the data (); +/// these tests read the texture back through the parity dump's own readback, inside a +/// frame, with sync + best-practices validation on. +/// +public class SsaoNoiseAlphaTests +{ + private const int NoiseSize = 16; + private const int GlRgba32f = 0x8814; // 34836, what both paths allocate + + private readonly ITestOutputHelper _output; + + public SsaoNoiseAlphaTests(ITestOutputHelper output) => _output = output; + + /// + /// The GL path's generation verbatim (ClientPlatformWindows.SetupDefaultFrameBuffers): + /// three floats per texel, uploaded as GL_RGB. + /// + private static float[] GlRgbNoise(Random random) + { + float[] rgb = new float[NoiseSize * NoiseSize * 3]; + Vec3f direction = new Vec3f(); + for (int texel = 0; texel < NoiseSize * NoiseSize; texel++) + { + direction.Set((float)random.NextDouble() * 2f - 1f, (float)random.NextDouble() * 2f - 1f, 0f).Normalize(); + rgb[texel * 3] = direction.X; + rgb[texel * 3 + 1] = direction.Y; + rgb[texel * 3 + 2] = direction.Z; + } + return rgb; + } + + /// + /// Same colour values and the same random draws as the GL path, alpha 1 where GL fills + /// it, and the stream position afterwards unchanged, so the kernel drawn next matches. + /// + [Fact] + public void NoiseMatchesTheGlUploadIncludingTheFilledAlpha() + { + var glRandom = new Random(5); + var deviceRandom = new Random(5); + float[] rgb = GlRgbNoise(glRandom); + float[] rgba = VulkanClientPlatform.BuildOptimumSsaoNoise(deviceRandom, NoiseSize); + + Assert.Equal(NoiseSize * NoiseSize * 4, rgba.Length); + for (int texel = 0; texel < NoiseSize * NoiseSize; texel++) + { + Assert.Equal(rgb[texel * 3], rgba[texel * 4]); + Assert.Equal(rgb[texel * 3 + 1], rgba[texel * 4 + 1]); + Assert.Equal(rgb[texel * 3 + 2], rgba[texel * 4 + 2]); + Assert.Equal(1f, rgba[texel * 4 + 3]); + } + Assert.Equal(glRandom.NextDouble(), deviceRandom.NextDouble()); + } + + /// + /// GPU readback through , the path that + /// writes 13-SSAO-color1-rgba32f.alpha.pfm. The platform's noise reads alpha 1.0 in all + /// 256 texels with the GL colour values; the pre-fix upload (identical colour, padding + /// alpha 0) on the same device reads 0.0, which is the Vulkan dump before the fix. + /// + [SkippableFact] + public void SsaoNoiseTextureReadsAlphaOneLikeOpenGl() + { + Skip.IfNot(GpuTest.TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + float[] noise = VulkanClientPlatform.BuildOptimumSsaoNoise(new Random(5), NoiseSize); + float[] preFix = (float[])noise.Clone(); + for (int texel = 0; texel < NoiseSize * NoiseSize; texel++) preFix[texel * 4 + 3] = 0f; + float[] glRgb = GlRgbNoise(new Random(5)); + + using (device) + { + VulkanDevice seam = device!; + int fixedTexture = Upload(seam, noise); + int preFixTexture = Upload(seam, preFix); + + seam.BeginFrame(); + OptimumTextureReadback? fixedReadback = seam.ReadTextureForParity(fixedTexture); + OptimumTextureReadback? preFixReadback = seam.ReadTextureForParity(preFixTexture); + seam.Present(); + GpuTest.AssertClean(seam); + + Assert.NotNull(fixedReadback); + Assert.NotNull(preFixReadback); + Assert.Equal(GlRgba32f, fixedReadback!.GlInternalFormat); + Assert.Equal(NoiseSize, fixedReadback.Width); + Assert.Equal(NoiseSize, fixedReadback.Height); + Assert.NotNull(fixedReadback.Floats); + Assert.NotNull(preFixReadback!.Floats); + + int alphaOne = 0; + int preFixAlphaZero = 0; + for (int texel = 0; texel < NoiseSize * NoiseSize; texel++) + { + Assert.Equal(glRgb[texel * 3], fixedReadback.Floats![texel * 4]); + Assert.Equal(glRgb[texel * 3 + 1], fixedReadback.Floats[texel * 4 + 1]); + Assert.Equal(glRgb[texel * 3 + 2], fixedReadback.Floats[texel * 4 + 2]); + if (fixedReadback.Floats[texel * 4 + 3] == 1f) alphaOne++; + if (preFixReadback.Floats![texel * 4 + 3] == 0f) preFixAlphaZero++; + } + _output.WriteLine("alpha 1.0 texels: " + alphaOne + "/256; pre-fix upload alpha 0.0 texels: " + preFixAlphaZero + "/256"); + Assert.Equal(NoiseSize * NoiseSize, preFixAlphaZero); + Assert.Equal(NoiseSize * NoiseSize, alphaOne); + } + } + + private static int Upload(VulkanDevice seam, float[] texels) + { + GCHandle handle = GCHandle.Alloc(texels, GCHandleType.Pinned); + try + { + return seam.CreateTexture2DRaw(NoiseSize, NoiseSize, GlRgba32f, handle.AddrOfPinnedObject(), 16); + } + finally + { + handle.Free(); + } + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/SsaoTemporalDitherTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +/// +/// GTAO roadmap step 2. Vanilla's ssao.fsh rotates its sample kernel with a Bayer-128 +/// dither locked to the screen grid: under a jittered camera every surface point draws +/// a different kernel every frame, and a temporal accumulator can only fight that, never +/// average it. Optimum's override advances the dither by the golden ratio per frame, +/// using the temporal pipeline's own frame index, so successive frames sample +/// complementary spiral directions. +/// +/// The two things worth pinning are both here, on the real shader, run on the device: +/// with the temporal pipeline on (TAAMOTION 1) the AO of a fixed scene changes between +/// consecutive frames, and with it off (TAAMOTION 0) it does not change at all - the +/// override has to be byte-identical to vanilla there, because a per-frame-varying +/// dither with nothing accumulating behind it is strictly worse than a fixed one. +/// +public class SsaoTemporalDitherTests(ITestOutputHelper output) +{ + private const int Size = 128; + private const int GlRgba32f = 0x8814; + private const int GlRgba8 = 0x8058; + private const int KernelSize = 64; + + /// Frame index the AO is rendered at, in order. The third repeats the first. + private static readonly float[] FrameIndices = [0f, 1f, 0f, 2f]; + + [SkippableTheory] + [InlineData(1)] + [InlineData(2)] + public void TheAoVariesPerFrameOnlyWhileTheTemporalPipelineIsOn(int quality) + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + + using (device) + { + VulkanDevice seam = device!; + var files = ShaderCorpus.LoadShaderFiles(); + + byte[][] withTemporal = RenderSequence(seam, files, quality, taaMotion: 1); + byte[][] withoutTemporal = RenderSequence(seam, files, quality, taaMotion: 0); + + // The scene really is occluded: several AO levels, not a flat 255. Printed + // because a scene that stops occluding would make every comparison below + // trivially pass with the temporal term removed. + var seen = new SortedSet(); + for (int i = 0; i < Size * Size; i++) seen.Add(withTemporal[0][i * 4]); + output.WriteLine("distinct AO values in frame 0: " + seen.Count + " -> " + string.Join(",", seen)); + Assert.True(seen.Count > 1, "the test scene produced no occlusion at all"); + + int changed = Differing(withTemporal[0], withTemporal[1]); + int repeated = Differing(withTemporal[0], withTemporal[2]); + int changedAgain = Differing(withTemporal[1], withTemporal[3]); + output.WriteLine($"SSAOLEVEL {quality}: temporal on - frame 0 vs 1: {changed} px differ, " + + $"1 vs 2: {changedAgain} px differ, frame 0 vs the frame that repeats index 0: {repeated} px differ"); + + // A different frame index really does rotate the kernel: a good share of the + // scene lands on a different occlusion value. The threshold is a tenth of the + // image, far below what the pass actually moves, so it fails on "the uniform + // is ignored", not on a driver's rounding. + Assert.True(changed > Size * Size / 10, $"AO did not vary between frames (only {changed} px)"); + Assert.True(changedAgain > Size * Size / 10, $"AO did not vary between frames (only {changedAgain} px)"); + // ... and the same index gives the same AO, so what varies is the frame index + // and not the device. + Assert.Equal(0, repeated); + + int off = Differing(withoutTemporal[0], withoutTemporal[1]); + int offAgain = Differing(withoutTemporal[1], withoutTemporal[3]); + output.WriteLine($"SSAOLEVEL {quality}: temporal off - frame 0 vs 1: {off} px differ, 1 vs 3: {offAgain} px differ"); + Assert.Equal(0, off); + Assert.Equal(0, offAgain); + + GpuTest.AssertClean(seam); + } + } + + /// + /// Renders the same scene once per entry in , each into its + /// own target, and reads them all back afterwards - the sequence has to complete before + /// any CPU wait, or the readback is what makes the frames differ. + /// + private static byte[][] RenderSequence( + VulkanDevice seam, Dictionary files, int quality, int taaMotion) + { + string defines = "#version 330 core\n#define SSAOLEVEL " + quality + "\n#define TAAMOTION " + taaMotion + "\n"; + int program = GpuTest.LinkProgram( + seam, + files["ssao.vsh"], + files["ssao.fsh"].Replace("#version 330 core", defines), + "ssao" + quality + "-taa" + taaMotion); + + (float[] positions, float[] normals) = Scene(); + int gPosition = UploadFloat(seam, positions); + int gNormal = UploadFloat(seam, normals); + int revealage = UploadOpaqueRed(seam); + // The pass never reads texNoise (the dither replaced it), but the sampler is + // declared, so it still needs something bound. + int noise = UploadFloat(seam, new float[Size * Size * 4]); + + int frames = FrameIndices.Length; + var targets = new int[frames]; + var colors = new int[frames]; + for (int i = 0; i < frames; i++) + { + targets[i] = seam.CreateFramebuffer(Size, Size); + colors[i] = seam.CreateTexture2DRaw(Size, Size, GlRgba8, IntPtr.Zero, 4); + seam.AttachTexture(targets[i], EnumFramebufferAttachment.ColorAttachment0, colors[i], 0); + } + + seam.SetViewport(0, 0, Size, Size); + seam.SetCullFace(false); + seam.SetDepthTest(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.SetSamplerUnit(program, "gPosition", 0); + seam.SetSamplerUnit(program, "gNormal", 1); + seam.SetSamplerUnit(program, "texNoise", 2); + seam.SetSamplerUnit(program, "revealage", 3); + seam.SetUniform(program, seam.GetUniformLocation(program, "screenSize"), (float)Size, (float)Size); + seam.SetUniformMatrix(program, seam.GetUniformLocation(program, "projection"), Projection()); + seam.SetUniformArray3(program, seam.GetUniformLocation(program, "samples"), KernelSize, Kernel()); + + // -1 when the shader was built without the temporal pipeline: the uniform is + // inside #if TAAMOTION == 1, exactly like the pass's own set is inside + // "if (OptimumConfig.EffectiveTaa)". + int frameIndexLocation = seam.GetUniformLocation(program, "temporalFrameIndex"); + Assert.Equal(taaMotion == 1, frameIndexLocation >= 0); + + for (int i = 0; i < frames; i++) + { + seam.BeginFrame(); + seam.BindFramebuffer(targets[i]); + seam.SetDrawBuffers(targets[i], 1); + seam.ClearColor(0, 0, 0, 0, 1); + seam.UseProgram(program); + if (frameIndexLocation >= 0) seam.SetUniform(program, frameIndexLocation, FrameIndices[i]); + seam.BindTexture(0, gPosition); + seam.BindTexture(1, gNormal); + seam.BindTexture(2, noise); + seam.BindTexture(3, revealage); + seam.DrawFullscreenTriangle(); + seam.Present(); + } + + var readback = new byte[frames][]; + seam.BeginFrame(); + for (int i = 0; i < frames; i++) readback[i] = seam.ReadBackLevel0ForTests(colors[i]); + seam.Present(); + return readback; + } + + /// + /// A view-space G-buffer the projection above reproduces exactly: a wall at 10 m + /// with 4x4 pixel blocks raised to 9.7 m, normals facing the camera. The raised + /// blocks put occluders within the shader's depth window all over the image, so + /// the AO of most pixels depends on which way the kernel points. + /// + private static (float[] Positions, float[] Normals) Scene() + { + var positions = new float[Size * Size * 4]; + var normals = new float[Size * Size * 4]; + for (int y = 0; y < Size; y++) + { + for (int x = 0; x < Size; x++) + { + bool raised = ((x / 4) + (y / 4)) % 2 == 0; + float distance = raised ? 9.7f : 10f; + float ndcX = (x + 0.5f) / Size * 2f - 1f; + float ndcY = (y + 0.5f) / Size * 2f - 1f; + int texel = (y * Size + x) * 4; + positions[texel] = ndcX * distance * TanHalfFov; + positions[texel + 1] = ndcY * distance * TanHalfFov; + positions[texel + 2] = -distance; + positions[texel + 3] = 0f; // attenuate + normals[texel + 2] = 1f; + normals[texel + 3] = 0f; // leavesHack off + } + } + return (positions, normals); + } + + private const float TanHalfFov = 0.7002075f; // tan(70 deg / 2) + + /// Column-major perspective, 70 degrees, square, near 0.1, far 100. + private static float[] Projection() + { + var m = new float[16]; + float f = 1f / TanHalfFov; + const float near = 0.1f, far = 100f; + m[0] = f; + m[5] = f; + m[10] = (far + near) / (near - far); + m[11] = -1f; + m[14] = 2f * far * near / (near - far); + return m; + } + + /// + /// A hemisphere kernel in the shape the client uploads: directions in the +z + /// hemisphere, pulled towards the origin quadratically so near samples dominate. + /// + private static float[] Kernel() + { + var random = new Random(7); + var kernel = new float[KernelSize * 3]; + for (int i = 0; i < KernelSize; i++) + { + double x = random.NextDouble() * 2.0 - 1.0; + double y = random.NextDouble() * 2.0 - 1.0; + double z = random.NextDouble(); + double length = Math.Sqrt(x * x + y * y + z * z); + double scale = 0.1 + 0.9 * ((double)i / KernelSize) * ((double)i / KernelSize); + kernel[i * 3] = (float)(x / length * scale); + kernel[i * 3 + 1] = (float)(y / length * scale); + kernel[i * 3 + 2] = (float)(z / length * scale); + } + return kernel; + } + + private static unsafe int UploadFloat(VulkanDevice seam, float[] texels) + { + fixed (float* data = texels) return seam.CreateTexture2DRaw(Size, Size, GlRgba32f, (IntPtr)data, 16); + } + + /// revealage = 1 everywhere, which is "nothing transparent here". + private static unsafe int UploadOpaqueRed(VulkanDevice seam) + { + var texels = new byte[Size * Size * 4]; + for (int i = 0; i < Size * Size; i++) texels[i * 4] = 255; + fixed (byte* data = texels) return seam.CreateTexture2DRaw(Size, Size, GlRgba8, (IntPtr)data, 4); + } + + private static int Differing(byte[] a, byte[] b) + { + Assert.Equal(a.Length, b.Length); + int count = 0; + for (int texel = 0; texel < a.Length / 4; texel++) + { + if (a[texel * 4] != b[texel * 4]) count++; + } + return count; + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/AmbientOcclusionTests.cs b/Optimum.Render.Vulkan.Tests/AmbientOcclusionTests.cs new file mode 100644 index 00000000..b22a30ce --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/AmbientOcclusionTests.cs @@ -0,0 +1,646 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Runtime.InteropServices; +using System.Text.RegularExpressions; +using Optimum.Render.Vulkan.AmbientOcclusion; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The GTAO visibility-bitmask pass (docs/vulkan.md#ambient-occlusion section C) on a device, +/// validation with sync and best practices on, against synthetic G-buffers drawn analytically: +/// a camera with a 90-degree square frustum looking down -z over a floor at y = -1, an optional +/// back wall, a one-texel slab and a hand-view rectangle. Depth goes through the real D32 +/// attachment, normals and the class channel through an RGBA16F one, exactly as Primary holds +/// them; the three compute passes run through as the platform does. +/// +/// The CPU facts at the bottom pin what the shaders and the C# side must agree on: the +/// reconstruction constants, the Hilbert table, the specialization ids and the push layout. +/// +public class AmbientOcclusionTests(ITestOutputHelper output) +{ + private const int Size = 64; + private const float Near = 0.1f; + private const float Far = 1000f; + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + // Ray per pixel in GL view space (tan = 1, so the ray is (ndc + jitter, -1)); the + // nearest analytic surface writes its GL depth, its GL view-space normal and class, and + // its view distance for the reconstruction check. + private const string SceneFragment = """ + #version 330 core + uniform int scene; + uniform float jitterX; + uniform float jitterY; + layout(location = 0) out vec4 outNormal; + layout(location = 1) out vec4 outViewDepth; + void main(void) + { + const float Near = 0.1; + const float Far = 1000.0; + vec2 ndc = gl_FragCoord.xy / 64.0 * 2.0 - 1.0; + vec3 ray = vec3(ndc.x + jitterX, ndc.y + jitterY, -1.0); + float t = 1e9; + vec3 n = vec3(0.0); + float surfaceClass = 0.0; + if (scene != 2 && ray.y < 0.0) + { + float floorT = -1.0 / ray.y; + if (floorT < t) { t = floorT; n = vec3(0.0, 1.0, 0.0); surfaceClass = 0.0; } + } + if (scene == 1 || scene == 3) + { + if (3.0 < t) { t = 3.0; n = vec3(0.0, 0.0, 1.0); surfaceClass = 0.0; } + } + if (scene == 2) + { + // A wall 4 blocks out and, 0.3 blocks in front of it, a rail one texel tall + // (a texel is 0.116 blocks at 3.7) flagged thin like foliage. + t = 4.0; n = vec3(0.0, 0.0, 1.0); surfaceClass = 0.0; + if (abs(ray.y * 3.7) < 0.058) { t = 3.7; surfaceClass = 1.0; } + } + if (scene == 3 && gl_FragCoord.x > 40.0 && gl_FragCoord.x < 56.0 && gl_FragCoord.y > 8.0 && gl_FragCoord.y < 24.0) + { + t = 0.5; n = vec3(0.0, 0.0, 1.0); surfaceClass = -1.0; + } + if (t > 1e8) + { + outNormal = vec4(0.0); + outViewDepth = vec4(0.0); + gl_FragDepth = 1.0; + return; + } + float a = -(Far + Near) / (Far - Near); + float b = -2.0 * Far * Near / (Far - Near); + float ndcZ = (a * -t + b) / t; + gl_FragDepth = ndcZ * 0.5 + 0.5; + outNormal = vec4(n, surfaceClass); + outViewDepth = vec4(t, 0.0, 0.0, 1.0); + } + """; + + private const string CopyFragment = """ + #version 330 core + uniform sampler2D ao; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = vec4(texelFetch(ao, ivec2(gl_FragCoord.xy), 0).rrr, 1.0); } + """; + + private enum Scene + { + OpenFloor = 0, + Crease = 1, + Slab = 2, + Hand = 3, + } + + private static float[] Projection(float jitterX = 0f, float jitterY = 0f) + { + var m = new float[16]; + m[0] = 1f; + m[5] = 1f; + m[8] = jitterX; + m[9] = jitterY; + m[10] = -(Far + Near) / (Far - Near); + m[11] = -1f; + m[14] = -2f * Far * Near / (Far - Near); + return m; + } + + /// The synthetic Primary: D32 depth, RGBA16F gNormal, an R32F view distance. + private sealed class Rig : IDisposable + { + public readonly VulkanDevice Device; + public readonly GtaoRenderer Ao; + public readonly int Program; + public readonly int Framebuffer; + public readonly int Depth; + public readonly int Normal; + public readonly int ViewDepth; + + public Rig(VulkanDevice device) + { + Device = device; + Program = GpuTest.LinkProgram(device, FullscreenVertex, SceneFragment, "ao-scene"); + Depth = device.CreateTexture2D(Size, Size, EnumTextureInternalFormat.DepthComponent32, + EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + Normal = device.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, + IntPtr.Zero, false); + ViewDepth = device.CreateTexture2DRaw(Size, Size, 0x822E, IntPtr.Zero, 4); + Framebuffer = device.CreateFramebuffer(Size, Size); + device.AttachTexture(Framebuffer, EnumFramebufferAttachment.ColorAttachment0, Normal, 0); + device.AttachTexture(Framebuffer, EnumFramebufferAttachment.ColorAttachment1, ViewDepth, 0); + device.AttachTexture(Framebuffer, EnumFramebufferAttachment.DepthAttachment, Depth, 0); + device.SetDrawBuffers(Framebuffer, 3); + Assert.True(device.CheckFramebufferComplete(Framebuffer, out string status), status); + Ao = new GtaoRenderer(device); + } + + public void Draw(Scene scene, float jitterX = 0f, float jitterY = 0f) + { + Device.BindFramebuffer(Framebuffer); + Device.SetViewport(0, 0, Size, Size); + Device.SetCullFace(false); + Device.SetBlend(false, EnumBlendMode.Standard); + Device.SetDepthTest(true); + Device.SetDepthMask(true); + Device.SetDepthFunc(0x0207); // GL_ALWAYS + Device.UseProgram(Program); + Device.SetUniform(Program, Device.GetUniformLocation(Program, "scene"), (int)scene); + Device.SetUniform(Program, Device.GetUniformLocation(Program, "jitterX"), jitterX); + Device.SetUniform(Program, Device.GetUniformLocation(Program, "jitterY"), jitterY); + Device.DrawFullscreenTriangle(); + } + + public int Render(GtaoSettings settings, uint noiseIndex, float[]? projection = null) + { + int texture = Ao.Render(Depth, Normal, projection ?? Projection(), settings, noiseIndex); + Assert.True(texture != 0, Ao.LastError); + return texture; + } + + /// The first channel of an 8-bit storage texture, in [0, 1], row 0 at the bottom. + public float[] ReadUnorm(int texture) => ReadUnormFrom(Device.ReadBackLevel0ForTests(texture), texture); + + public float[] ReadUnormFrom(byte[] bytes, int texture) + { + int channels = Device.TextureOf(texture)!.Format switch + { + Format.R8Unorm => 1, + Format.R8G8Unorm => 2, + _ => 4, + }; + var values = new float[Size * Size]; + for (int i = 0; i < values.Length; i++) values[i] = bytes[i * channels] / 255f; + return values; + } + + public byte ByteAt(byte[] bytes, int texture, int x, int y) + { + int channels = Device.TextureOf(texture)!.Format switch + { + Format.R8Unorm => 1, + Format.R8G8Unorm => 2, + _ => 4, + }; + return bytes[(y * Size + x) * channels]; + } + + public void Dispose() => Ao.Dispose(); + } + + private static double Mean(float[] values, Func inside) + { + double sum = 0; + int count = 0; + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + if (!inside(x, y)) continue; + sum += values[y * Size + x]; + count++; + } + Assert.True(count > 0); + return sum / count; + } + + private static GtaoSettings Settings(GtaoIntegration integration = GtaoIntegration.BitmaskCosine, + GtaoPreset preset = GtaoPreset.High) => + GtaoSettings.ForPreset(preset, temporal: true) with { Integration = integration }; + + [SkippableFact] + public void AnOpenPlaneIsUnoccludedAndACreaseIsDarker() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + + device!.BeginFrame(); + rig.Draw(Scene.OpenFloor); + float[] open = rig.ReadUnorm(rig.Render(Settings(), 0)); + device.Present(); + + device.BeginFrame(); + rig.Draw(Scene.Crease); + float[] crease = rig.ReadUnorm(rig.Render(Settings(), 0)); + device.Present(); + + // Floor rows 0..20 are 1.0 to 2.9 blocks away: open ground with nothing above it. + double openMin = 1.0; + for (int y = 0; y <= 20; y++) + for (int x = 0; x < Size; x++) + openMin = Math.Min(openMin, open[y * Size + x]); + output.WriteLine("open floor min visibility " + openMin.ToString("F4")); + Assert.True(openMin >= 0.98, "open floor min visibility " + openMin); + + // The wall meets the floor at row 21.3; the rows just below it sit in the crease. + double inCrease = Mean(crease, (_, y) => y is >= 17 and <= 20); + double awayFromCrease = Mean(crease, (_, y) => y is >= 0 and <= 4); + output.WriteLine("crease " + inCrease.ToString("F4") + ", away " + awayFromCrease.ToString("F4")); + Assert.True(inCrease < 0.9, "crease visibility " + inCrease); + Assert.True(inCrease < awayFromCrease - 0.05, "crease " + inCrease + " vs away " + awayFromCrease); + + rig.Dispose(); + GpuTest.AssertClean(device); + } + } + + [SkippableFact] + public void AOneTexelSlabOccludesLessWithTheBitmaskThanWithTheHorizonIntegral() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + // The wall rows within the effect radius (about 9 texels) of the rail, the rail excluded. + bool AroundTheSlab(int x, int y) => x is >= 8 and < 56 && y is >= 26 and <= 38 && y is < 31 or > 33; + + var means = new Dictionary(); + foreach (GtaoIntegration integration in new[] { GtaoIntegration.BitmaskCosine, GtaoIntegration.Horizon }) + { + device!.BeginFrame(); + rig.Draw(Scene.Slab); + means[integration] = Mean(rig.ReadUnorm(rig.Render(Settings(integration), 0)), AroundTheSlab); + device.Present(); + output.WriteLine(integration + " mean visibility around the slab " + means[integration].ToString("F4")); + } + + // The horizon integral treats the slab as infinitely thick; the bitmask lets light + // pass behind its thickness. + Assert.True(means[GtaoIntegration.Horizon] < 0.97, "the slab must visibly occlude the horizon variant"); + Assert.True(means[GtaoIntegration.BitmaskCosine] > means[GtaoIntegration.Horizon] + 0.02, + "bitmask " + means[GtaoIntegration.BitmaskCosine] + " vs horizon " + means[GtaoIntegration.Horizon]); + + rig.Dispose(); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void CosineWeightingCountsGrazingOccludersLessThanUniformSectors() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + bool NearTheCrease(int _, int y) => y is >= 14 and <= 20; + + var means = new Dictionary(); + foreach (GtaoIntegration integration in new[] { GtaoIntegration.BitmaskCosine, GtaoIntegration.BitmaskUniform }) + { + device!.BeginFrame(); + rig.Draw(Scene.Crease); + means[integration] = Mean(rig.ReadUnorm(rig.Render(Settings(integration), 0)), NearTheCrease); + device.Present(); + output.WriteLine(integration + " mean visibility near the crease " + means[integration].ToString("F4")); + } + + // A crease occludes the floor from its horizon up; cosine weighting gives the + // near-horizon sectors less weight ((1 - cos a) / 2 against a / pi of the slice for + // an occluder up to elevation a), so the documented direction is brighter. + Assert.True(means[GtaoIntegration.BitmaskUniform] < 0.99); + Assert.True(means[GtaoIntegration.BitmaskCosine] > means[GtaoIntegration.BitmaskUniform], + "cosine " + means[GtaoIntegration.BitmaskCosine] + " vs uniform " + means[GtaoIntegration.BitmaskUniform]); + + rig.Dispose(); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void TheHandClassAndTheSkyReceiveNoOcclusion() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + + device!.BeginFrame(); + rig.Draw(Scene.Hand); + int texture = rig.Render(Settings(), 0); + byte[] hand = device.ReadBackLevel0ForTests(texture); + byte[] handWorking = device.ReadBackLevel0ForTests(rig.Ao.WorkingTermTexture); + device.Present(); + + int handPixels = 0; + for (int y = 9; y <= 23; y++) + for (int x = 41; x <= 55; x++) + { + Assert.Equal(255, rig.ByteAt(hand, texture, x, y)); + Assert.Equal(170, rig.ByteAt(handWorking, rig.Ao.WorkingTermTexture, x, y)); // 1 / 1.5 + handPixels++; + } + // And the crease around the hand is still occluded: the pass did run. + Assert.True(Mean(rig.ReadUnormFrom(hand, texture), (x, y) => x < 36 && y is >= 17 and <= 20) < 0.95); + + device.BeginFrame(); + rig.Draw(Scene.OpenFloor); + texture = rig.Render(Settings(), 0); + byte[] sky = device.ReadBackLevel0ForTests(texture); + device.Present(); + for (int y = 32; y < Size; y++) + for (int x = 0; x < Size; x++) + Assert.Equal(255, rig.ByteAt(sky, texture, x, y)); + + output.WriteLine(handPixels + " hand pixels and " + (Size - 32) * Size + " sky pixels at visibility 1"); + rig.Dispose(); + GpuTest.AssertClean(device); + } + } + + [SkippableFact] + public void TheWorkingDepthIsFiniteAndTheVisibilityStaysInItsRange() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + foreach (GtaoIntegration integration in Enum.GetValues()) + { + device!.BeginFrame(); + rig.Draw(Scene.Hand); + float[] visibility = rig.ReadUnorm(rig.Render(Settings(integration), 3)); + for (uint level = 0; level < GtaoRenderer.DepthLevels; level++) + { + float[] depth = MemoryMarshal.Cast(device.ReadBackLevelForTests(rig.Ao.WorkingDepthTexture, level)).ToArray(); + Assert.Equal((Size >> (int)level) * (Size >> (int)level), depth.Length); + foreach (float z in depth) Assert.True(float.IsFinite(z) && z > 0f && z <= Far * 1.001f, "level " + level + " depth " + z); + } + // max(0.03, v) survives the UNORM packing; a NaN would have stored 0. + foreach (float v in visibility) Assert.InRange(v, 0.03f - 1f / 255f, 1f); + device.Present(); + } + rig.Dispose(); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void TheResultIsStableAcrossPresentedFramesWithAFixedNoiseIndex() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + const int frames = 6; + int copy = GpuTest.LinkProgram(device!, FullscreenVertex, CopyFragment, "ao-copy"); + device!.SetSamplerUnit(copy, "ao", 0); + var colours = new int[frames]; + var targets = new int[frames]; + for (int i = 0; i < frames; i++) + { + colours[i] = device.CreateTexture2DRaw(Size, Size, 0x8058, IntPtr.Zero, 4); + targets[i] = device.CreateFramebuffer(Size, Size); + device.AttachTexture(targets[i], EnumFramebufferAttachment.ColorAttachment0, colours[i], 0); + device.SetDrawBuffers(targets[i], 1); + } + + // No readback in the loop: each frame draws the scene, runs the passes and copies + // the output into its own target, then presents. + for (int i = 0; i < frames; i++) + { + device.BeginFrame(); + rig.Draw(Scene.Crease); + int texture = rig.Render(Settings(), 11); + device.BindFramebuffer(targets[i]); + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(false); + device.UseProgram(copy); + device.BindTexture(0, texture); + device.DrawFullscreenTriangle(); + device.BindTexture(0, 0); + device.Present(); + } + + device.BeginFrame(); + byte[] first = device.ReadBackLevel0ForTests(colours[0]); + Assert.Contains(first.Where((_, i) => i % 4 == 0), b => b < 250); + for (int i = 1; i < frames; i++) Assert.Equal(first, device.ReadBackLevel0ForTests(colours[i])); + device.Present(); + + // The noise is live: another index moves the pre-denoise term. + device.BeginFrame(); + rig.Draw(Scene.Crease); + rig.Render(Settings(), 11); + byte[] still = device.ReadBackLevel0ForTests(rig.Ao.WorkingTermTexture); + rig.Render(Settings(), 12); + byte[] moved = device.ReadBackLevel0ForTests(rig.Ao.WorkingTermTexture); + device.Present(); + Assert.NotEqual(still, moved); + + rig.Dispose(); + GpuTest.AssertClean(device); + } + } + + [SkippableFact] + public void TheWorkingDepthReconstructsTheAnalyticViewDepthWithinATenthOfAPercent() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + var rig = new Rig(device!); + // A TAA-sized jitter, so the jitter columns of the projection are exercised too. + const float jitterX = 0.6f / Size, jitterY = -0.35f / Size; + + device!.BeginFrame(); + rig.Draw(Scene.Crease, jitterX, jitterY); + rig.Render(Settings(), 0, Projection(jitterX, jitterY)); + float[] working = MemoryMarshal.Cast(device.ReadBackLevelForTests(rig.Ao.WorkingDepthTexture, 0)).ToArray(); + float[] analytic = MemoryMarshal.Cast(device.ReadBackLevel0ForTests(rig.ViewDepth)).ToArray(); + device.Present(); + + int compared = 0; + double worst = 0; + for (int i = 0; i < analytic.Length; i++) + { + if (analytic[i] <= 0f || analytic[i] >= 100f) continue; + double error = Math.Abs(working[i] - analytic[i]) / analytic[i]; + worst = Math.Max(worst, error); + compared++; + } + output.WriteLine(compared + " pixels, worst relative view-depth error " + worst.ToString("E3")); + Assert.True(compared > Size * Size / 2); + Assert.True(worst < 1e-3, "worst relative error " + worst); + + rig.Dispose(); + GpuTest.AssertClean(device); + } + } + + [SkippableFact] + public void EveryStorageFormatFallbackCompilesWithMatchingQualifiers() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + // StorageFormats' candidate chains end in RGBA32F for the depth and RGBA8 for the terms. + foreach (Format depth in new[] { Format.R32Sfloat, Format.R32G32B32A32Sfloat }) + foreach (Format term in new[] { Format.R8Unorm, Format.R8G8Unorm, Format.R8G8B8A8Unorm }) + foreach (string file in new[] { "prefilter.comp", "main.comp", "denoise.comp" }) + { + string source = GtaoShaderSources.Build(file, depth, term); + int program = device!.CreateComputeProgram(source, file, Array.Empty(), GtaoSettings.PushConstantBytes); + Assert.True(program > 0, file + " " + depth + "/" + term + ": " + device.GetError()); + device.DeleteComputeProgram(program); + } + GpuTest.AssertClean(device!); + } + } + + // ------------------------------------------------------------------ device-free + + [Fact] + public void TheReconstructionConstantsInvertAJitteredGlProjection() + { + float[] m = Projection(0.6f / Size, -0.35f / Size); + GtaoProjection projection = GtaoProjection.From(m)!.Value; + var random = new Random(7); + for (int i = 0; i < 200; i++) + { + float z = 0.2f + (float)random.NextDouble() * 150f; + float x = ((float)random.NextDouble() * 2f - 1f) * z; + float y = ((float)random.NextDouble() * 2f - 1f) * z; + // GL: clip = P * (x, y, -z, 1) + float clipX = m[0] * x + m[8] * -z; + float clipY = m[5] * y + m[9] * -z; + float clipZ = m[10] * -z + m[14]; + float clipW = -(-z); + float u = (clipX / clipW + 1f) / 2f; + float v = (clipY / clipW + 1f) / 2f; + float depth = (clipZ / clipW + 1f) / 2f; + + float viewDepth = projection.ViewDepth(depth); + Assert.True(Math.Abs(viewDepth - z) / z < 1e-3, "depth " + viewDepth + " vs " + z); + (float rx, float ry, _) = projection.ViewPosition(u, v, z); + Assert.True(Math.Abs(rx - x) <= 1e-3 * z, "x " + rx + " vs " + x); + Assert.True(Math.Abs(ry - y) <= 1e-3 * z, "y " + ry + " vs " + y); + } + Assert.Null(GtaoProjection.From(new float[16])); + } + + [Fact] + public void TheHilbertTableIsAPermutationWithAdjacentNeighbours() + { + float[] table = HilbertLut.Build(); + Assert.Equal(4096, table.Length); + var positions = new (int X, int Y)[4096]; + var seen = new bool[4096]; + for (int y = 0; y < 64; y++) + for (int x = 0; x < 64; x++) + { + int index = (int)table[y * 64 + x]; + Assert.False(seen[index]); + seen[index] = true; + positions[index] = (x, y); + } + for (int i = 1; i < 4096; i++) + { + Assert.Equal(1, Math.Abs(positions[i].X - positions[i - 1].X) + Math.Abs(positions[i].Y - positions[i - 1].Y)); + } + } + + [Fact] + public void TheSpecializationIdsAndThePushBlockAgreeWithTheInclude() + { + string include = File.ReadAllText(Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk", "gtao", "common.glsl")); + var ids = Regex.Matches(include, @"#define GTAO_SPEC_(\w+) (\d+)") + .ToDictionary(m => m.Groups[1].Value.Replace("_", ""), m => int.Parse(m.Groups[2].Value), StringComparer.OrdinalIgnoreCase); + foreach (var field in typeof(GtaoSpecialization).GetFields().Where(f => f.IsLiteral && f.Name != "Count")) + { + Assert.True(ids.TryGetValue(field.Name, out int id), field.Name + " missing from common.glsl"); + Assert.Equal((int)field.GetValue(null)!, id); + } + Assert.Equal(ids.Count, typeof(GtaoSpecialization).GetFields().Count(f => f.IsLiteral && f.Name != "Count")); + Assert.Equal(ids.Values.Max() + 1, GtaoSpecialization.Count); + + // Every member four bytes wide, in the order GtaoSettings.PushConstants writes them. + string block = include[include.IndexOf("uniform GtaoConstants", StringComparison.Ordinal)..]; + block = block[..block.IndexOf("} gtao;", StringComparison.Ordinal)]; + string[] members = Regex.Matches(block, @"^\s*(float|uint) (\w+);", RegexOptions.Multiline).Select(m => m.Groups[2].Value).ToArray(); + Assert.Equal(new[] + { + "depthUnpackMul", "depthUnpackAdd", "ndcToViewMulX", "ndcToViewMulY", "ndcToViewAddX", "ndcToViewAddY", + "effectRadius", "effectFalloffRange", "radiusMultiplier", "finalValuePower", "sampleDistributionPower", + "depthMipSamplingOffset", "thickness", "thicknessThin", "thicknessDistanceScale", "farFadeBias", "farFadeScale", + "noiseIndex", "denoiseBlurBeta", "reserved", + }, members); + Assert.Equal(GtaoSettings.PushConstantBytes, members.Length * 4); + + byte[] push = new GtaoSettings { EffectRadius = 0.9f }.PushConstants(GtaoProjection.From(Projection())!.Value, 42); + Assert.Equal(GtaoSettings.PushConstantBytes, push.Length); + Assert.Equal(0.9f, BitConverter.ToSingle(push, 24)); + Assert.Equal(42u, BitConverter.ToUInt32(push, 68)); + } + + [Fact] + public void PresetsVariantsAndTheAlbedoHookResolveAsDocumented() + { + Assert.Equal((1u, 2u, 1u), Counts(GtaoSettings.ForPreset(GtaoPreset.Low, true))); + Assert.Equal((2u, 2u, 1u), Counts(GtaoSettings.ForPreset(GtaoPreset.Medium, true))); + Assert.Equal((3u, 3u, 1u), Counts(GtaoSettings.ForPreset(GtaoPreset.High, true))); + Assert.Equal((9u, 3u, 2u), Counts(GtaoSettings.ForPreset(GtaoPreset.Ultra, true))); + Assert.Equal((2u, 2u, 2u), Counts(GtaoSettings.ForPreset(GtaoPreset.Medium, false))); + Assert.Equal((1u, 2u, 2u), Counts(GtaoSettings.ForStableTemporal(GtaoPreset.Low))); + Assert.Equal((3u, 3u, 2u), Counts(GtaoSettings.ForStableTemporal(GtaoPreset.Medium))); + Assert.Equal((3u, 3u, 2u), Counts(GtaoSettings.ForStableTemporal(GtaoPreset.High))); + Assert.Equal((9u, 3u, 2u), Counts(GtaoSettings.ForStableTemporal(GtaoPreset.Ultra))); + Assert.Equal(GtaoPreset.Medium, GtaoSettings.ParsePreset("nonsense")); + Assert.Equal(GtaoPreset.Ultra, GtaoSettings.ParsePreset(" Ultra ")); + + GtaoSettings defaults = GtaoSettings.ForPreset(GtaoPreset.Medium, true); + Assert.Equal(GtaoIntegration.BitmaskCosine, defaults.Integration); + Assert.Equal(GtaoThickness.Random, defaults.Thickness); + Assert.True(defaults.ClassChannel); + Assert.Equal(64u, defaults.NoiseCycle); + Assert.Equal(1.0f, defaults.FinalValuePower); // no FinalValuePower (C.11) + + var environment = new Dictionary + { + ["OPTIMUM_AO_INTEGRATION"] = "horizon", + ["OPTIMUM_AO_THICKNESS"] = "const", + ["OPTIMUM_AO_CLASS_CHANNEL"] = "0", + ["OPTIMUM_AO_NOISE_CYCLE"] = "61", + ["OPTIMUM_AO_DENOISE_PASSES"] = "3", + ["OPTIMUM_AO_NORMAL_EDGES"] = "1", + ["OPTIMUM_AO_TONE"] = "multibounce", + ["OPTIMUM_AO_FINAL_POWER"] = "2.2", + }; + GtaoSettings measured = defaults.WithEnvironment(name => environment.GetValueOrDefault(name)); + uint[] specialization = measured.MainSpecialization(); + Assert.Equal((uint)GtaoIntegration.Horizon, specialization[GtaoSpecialization.Integration]); + Assert.Equal((uint)GtaoThickness.Constant, specialization[GtaoSpecialization.Thickness]); + Assert.Equal(0u, specialization[GtaoSpecialization.ClassChannel]); + Assert.Equal(61u, specialization[GtaoSpecialization.NoiseCycle]); + Assert.Equal(1u, specialization[GtaoSpecialization.NormalEdges]); + Assert.Equal(3u, measured.DenoisePasses); + Assert.Equal(2.2f, measured.FinalValuePower); + Assert.Equal(defaults, defaults.WithEnvironment(_ => "garbage")); + + // Multi-bounce needs a real albedo; the lit scene colour is not one. + Assert.Equal(GtaoTone.Linear, measured.EffectiveTone(0, out string? refusal)); + Assert.NotNull(refusal); + Assert.Equal(GtaoTone.MultiBounce, measured.EffectiveTone(123, out refusal)); + Assert.Null(refusal); + Assert.Equal(0u, GtaoSettings.DenoiseSpecialization(false)[GtaoSpecialization.FinalApply]); + Assert.Equal(1u, GtaoSettings.DenoiseSpecialization(true)[GtaoSpecialization.FinalApply]); + + static (uint, uint, uint) Counts(GtaoSettings s) => (s.SliceCount, s.StepsPerSlice, s.DenoisePasses); + } +} diff --git a/Optimum.Render.Vulkan.Tests/AssemblyInfo.cs b/Optimum.Render.Vulkan.Tests/AssemblyInfo.cs new file mode 100644 index 00000000..7fcf449b --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/AssemblyInfo.cs @@ -0,0 +1,59 @@ +using System; +using System.Runtime.CompilerServices; +using Xunit; + +// These tests drive a real GPU driver, and three of them additionally drive +// GLFW's process-global init and terminate. Neither is safe to do from several +// threads at once: xunit's default of running collections in parallel crashed +// the test host outright once the windowing tests joined the suite. +// +// Some integration and stress cases are long-running. Keep real-device tests +// serialized until the GLFW and device lifecycle are isolated from pure cases. +[assembly: CollectionBehavior(DisableTestParallelization = true)] + +namespace Optimum.Render.Vulkan.Tests; + +internal static class TestEnvironment +{ + /// + /// Keeps implicit Vulkan layers out of the test process. + /// + /// Overlays a developer happens to have installed - MangoHud, gamescope's + /// WSI layer, vendor layers - are loaded into every Vulkan instance on the + /// machine, and they can produce validation errors of their own. The + /// gamescope layer on this machine enables VK_KHR_present_mode_fifo_latest_ready + /// without the VK_KHR_swapchain it depends on, which fails the validation + /// assertions in twenty-five tests for a reason that has nothing to do with + /// this backend. Excluding them makes the suite depend only on our own calls. + /// + /// Set before any Vulkan call, because the loader reads it when an instance + /// is created. + /// + [ModuleInitializer] + internal static void DisableImplicitLayers() + { + if (Environment.GetEnvironmentVariable("VK_LOADER_LAYERS_DISABLE") != null) return; + + Environment.SetEnvironmentVariable("VK_LOADER_LAYERS_DISABLE", "~implicit~"); + + // .NET's SetEnvironmentVariable only updates the managed copy on Unix; + // the Vulkan loader is native and reads the real environment, so it has + // to be set through libc as well or nothing changes. + if (OperatingSystem.IsLinux() || OperatingSystem.IsMacOS()) + { + try + { + SetNativeEnvironmentVariable("VK_LOADER_LAYERS_DISABLE", "~implicit~", 1); + } + catch (DllNotFoundException) + { + } + catch (EntryPointNotFoundException) + { + } + } + } + + [System.Runtime.InteropServices.DllImport("libc", EntryPoint = "setenv")] + private static extern int SetNativeEnvironmentVariable(string name, string value, int overwrite); +} diff --git a/Optimum.Render.Vulkan.Tests/ChunkTerrainRenderTests.cs b/Optimum.Render.Vulkan.Tests/ChunkTerrainRenderTests.cs new file mode 100644 index 00000000..6df17d3d --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ChunkTerrainRenderTests.cs @@ -0,0 +1,1052 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// Terrain drawn with the game's own chunk shader, through the seam and nothing +/// else. +/// +/// Everything else about chunks in this project stops short of a draw: the corpus +/// proves the shaders become SPIR-V, and ChunkRenderPathTests proves they become +/// pipelines. Neither puts a block on screen. These build the vertex data a +/// tesselated chunk actually carries - positions, UVs, per-vertex colour and the +/// packed render-flags word, each in its own buffer, exactly as +/// AllocateEmptyMesh lays them out - bind the real chunkopaque program with its +/// real samplers and matrices, draw, and read the pixels back. +/// +/// A world in the client needs a signed-in account, so this is terrain rendered +/// through the backend without one: real shader, real vertex format, real device +/// path, real pixels. +/// +public class ChunkTerrainRenderTests +{ + private readonly ITestOutputHelper _output; + + public ChunkTerrainRenderTests(ITestOutputHelper output) => _output = output; + + private const int Size = 64; + private const int UpNormalFlags = 7 << 18; + + private static float[] CreateFaceRecord() + { + var record = new float[16]; + // unpackNormal normalizes the packed vector. An all-zero flags word + // would normalize (0,0,0), producing NaNs on some drivers. + Array.Fill(record, BitConverter.Int32BitsToSingle(UpNormalFlags), 8, 4); + return record; + } + + private static bool TryCreateDevice(ITestOutputHelper output, out VulkanDevice? device) => + GpuTest.TryCreateDevice(output, out device); + + private sealed class Shader : IShader + { + public EnumShaderType Type { get; set; } + public string Code { get; set; } = ""; + public string PrefixCode { get; set; } = ""; + public bool Compile() => true; + } + + private sealed class Program : IShaderProgram + { + public int ProgramId { get; set; } + public string AssetDomain { get; set; } = "game"; + public int PassId { get; set; } + public string PassName { get; set; } = "chunkopaque"; + public bool ClampTexturesToEdge { get; set; } + public IShader VertexShader { get; set; } = null!; + public IShader FragmentShader { get; set; } = null!; + public IShader GeometryShader { get; set; } = null!; + public bool Oit { get; set; } = true; + public bool Disposed => false; + public bool LoadError => false; + public Vintagestory.API.Datastructures.OrderedDictionary UBOs { get; } = new(); + public bool Compile() => true; + public bool HasUniform(string uniformName) => false; + public void Use() { } + public void Stop() { } + public void Dispose() { } + public void Uniform(string uniformName, float value) { } + public void Uniform(string uniformName, int value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2f value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2i value) { } + public void Uniform(string uniformName, float valueX, float valueY) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec3f value) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ, float valueW) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec4f value) { } + public void Uniforms4(string uniformName, int count, float[] values) { } + public void UniformMatrix(string uniformName, float[] matrix) { } + public void BindTexture2D(string samplerName, int textureId, int textureNumber) { } + public void BindTextureCube(string samplerName, int textureId, int textureNumber) { } + public void UniformMatrices(string uniformName, int count, float[] matrix) { } + public void UniformMatrices4x3(string uniformName, int count, float[] matrix) { } + } + + /// + /// One tesselated block face, in the layout the chunk tesselator emits. + /// + /// Positions are chunk-local, UVs index the block atlas, the colour carries + /// baked light, and the flags word packs glow, z-offset, waving bits and the + /// normal - the field whose undefined value made the whole interface vanish + /// when vertex-attribute defaults were missing. + /// + private static MeshData BuildBlockFace() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + + // A quad covering the middle of the viewport in clip space, so the draw + // is checkable by reading the centre pixel. + float[] positions = + { + -0.5f, -0.5f, 0f, + 0.5f, -0.5f, 0f, + 0.5f, 0.5f, 0f, + -0.5f, 0.5f, 0f, + }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags( + positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], + Vintagestory.API.MathTools.ColorUtil.WhiteArgb, + // Normal pointing up, no glow, no waving: the flags word a solid + // top face carries. + flags: UpNormalFlags); + } + + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) + { + mesh.AddIndex(index); + } + return mesh; + } + + [SkippableTheory] + [InlineData(false)] + [InlineData(true)] + public unsafe void TheTopsoilShaderSamplesTheGrassTileWithPackedSecondaryUvs(bool ssbo) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + var variant = ShaderCorpus.Variants().First(); + variant.UseSsbo = ssbo ? 1 : 0; + int program = LinkFromCorpus(seam, ShaderCorpus.BuildProgram("chunktopsoil", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant), "chunktopsoil"); + int unit = BindEveryDeclaredSampler(device!, seam, program); + + // The top grass texture occupies tile (2,1), one tile to the right + // of its packed UV origin, as in the real topsoil atlas. Signed + // normalization doubles the UVs and sends them into a red tile. + var atlasPixels = new byte[4 * 4 * 4]; + for (int i = 0; i < 16; i++) + { + atlasPixels[i * 4] = 255; + atlasPixels[i * 4 + 3] = 255; + } + atlasPixels[6 * 4] = 0; + atlasPixels[6 * 4 + 1] = 255; + int atlas; + fixed (byte* pixels = atlasPixels) + atlas = seam.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + foreach (string name in new[] { "terrainTex", "terrainTexLinear" }) + { + seam.SetSamplerUnit(program, name, unit); + seam.BindTexture(unit++, atlas); + } + + MeshData face = BuildBlockFace(); + face.CustomShorts = new CustomMeshDataPartShort(8) + { + InterleaveSizes = new[] { 2 }, InterleaveOffsets = new[] { 0 }, + InterleaveStride = 4, Conversion = DataConversion.NormalizedFloat, + }; + for (int i = 0; i < 4; i++) + face.CustomShorts.AddPackedUV(0.375f, 0.375f, isU2: false, isV2: false); + + int mesh = seam.CreateEmptyMesh(48, 0, 32, 16, 16, 24, + null, face.CustomShorts, null, null, EnumDrawMode.Triangles, false, ssbo); + seam.UpdateMesh(mesh, face); + if (ssbo) + { + var record = CreateFaceRecord(); + record[0] = -0.5f; record[1] = -0.5f; + record[4] = 0.5f; record[13] = 0.5f; + fixed (float* pixels = record) + seam.UpdateMeshStorageBuffer(mesh, (IntPtr)pixels, 0, 64); + } + + int target = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.SetDrawBuffers(framebuffer, 1); + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 1f, 0f, 1f, 1f); + seam.UseProgram(program); + SetIdentityMatrices(seam, program); + SetViewUniforms(seam, program); + seam.SetUniform(program, seam.GetUniformLocation(program, "blockTextureSize"), 0.25f, 0.25f); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetCullFace(true); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo); + seam.Present(); + + byte[] output = ReadTarget(seam, framebuffer); + int centre = (Size / 2 * Size + Size / 2) * 4; + Assert.True(output[centre + 1] > 20 && output[centre] < 5 && output[centre + 2] < 5, + $"Expected grass green; got {output[centre]}, {output[centre + 1]}, {output[centre + 2]}"); + Assert.True(IsClearColour(output, 0), "The pooled face extended outside its geometry."); + AssertClean(seam); + } + } + + /// + /// The real chunkopaque program, drawing a real tesselated face, checked by + /// reading the pixels back. + /// + [SkippableFact] + public unsafe void TheRealChunkShaderDrawsTesselatedTerrain() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + ShaderCorpus.ShaderVariant variant = ShaderCorpus.Variants().First(); + + List stages = + ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkopaque"); + + // Every sampler the program declares needs something bound, or the + // draw is skipped rather than drawn - the descriptor would be + // incomplete. A single white texel stands in for the block atlas. + BindEveryDeclaredSampler(device!, seam, programId); + + int target = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + Assert.True(seam.CheckFramebufferComplete(framebuffer, out string status), status); + + int mesh = seam.CreateMesh(BuildBlockFace(), staticDraw: true); + Assert.True(mesh > 0, seam.GetError() ?? "mesh upload failed"); + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + // Cleared to magenta rather than black, because chunkopaque's output + // depends on lighting, fog and atlas contents that are all zero here - + // it legitimately shades to black. Testing "the pixel is lit" would + // then be indistinguishable from "nothing drew". Testing "the pixel + // changed" detects rasterisation whatever the shader decides to emit. + seam.ClearColor(0, 1f, 0f, 1f, 1f); + seam.ClearDepth(1f); + + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(true); + seam.SetDepthFunc(0x203); // GL_LEQUAL + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + + seam.DrawMesh(mesh); + seam.Present(); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.BindFramebuffer(framebuffer); + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + + int centre = (Size / 2 * Size + Size / 2) * 4; + int corner = (2 * Size + 2) * 4; + + _output.WriteLine($"centre RGBA = {pixels[centre]}, {pixels[centre + 1]}, " + + $"{pixels[centre + 2]}, {pixels[centre + 3]}"); + _output.WriteLine($"corner RGBA = {pixels[corner]}, {pixels[corner + 1]}, " + + $"{pixels[corner + 2]}, {pixels[corner + 3]}"); + + // The face covers the middle and nothing else: geometry actually + // rasterised, in the right place, and did not cover the whole target. + bool centreChanged = !IsClearColour(pixels, centre); + bool cornerUntouched = IsClearColour(pixels, corner); + + Assert.True(centreChanged, "the block face did not rasterise"); + Assert.True(cornerUntouched, "the block face covered the whole target"); + + AssertClean(seam); + } + } + + /// + /// The same face through the shadow-map program, which renders depth only and + /// is the pass every shadowed chunk goes through first. + /// + [SkippableFact] + public void TheShadowMapProgramDrawsTerrainIntoADepthOnlyTarget() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + + List stages = ShaderCorpus.BuildProgram( + "chunkshadowmap", files, includes, ShaderCorpus.Variants().First()); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkshadowmap"); + BindEveryDeclaredSampler(device!, seam, programId); + + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0); + + int mesh = seam.CreateMesh(BuildBlockFace(), staticDraw: true); + Assert.True(mesh > 0, seam.GetError() ?? "mesh upload failed"); + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearDepth(1f); + + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(true); + seam.SetDepthMask(true); + seam.SetDepthFunc(0x203); + seam.SetCullFace(false); + + seam.DrawMesh(mesh); + seam.Present(); + + // A depth-only pass has nothing to read back as colour; what matters + // is that it recorded and submitted without the device refusing the + // draw or the validation layer objecting. + AssertClean(seam); + } + } + + // ------------------------------------------------------------------ helpers + + /// + /// The path the world actually renders through: an SSBO pool, one packed + /// face record in its storage slot, the fixed quad index pattern, the storage + /// descriptor set, and culling on. The attribute-variant test above covers + /// none of that, which is how a world of shards got past the suite. + /// + /// The record follows the shader's std430 FaceData - xyz, uv, xyzA, uvSize, + /// flags[4], xyzB, colormapData, 64 bytes - and the decode is + /// xyz + ((v+1)&2)*xyzA + (v&2)*xyzB, so xyzA and xyzB are half the + /// quad's two edges. The flags encode an upward normal; their integer bits + /// are preserved in the float array used to upload the record. + /// + [SkippableFact] + public unsafe void TheSsboChunkPathDrawsAFaceFromAPackedRecordWithCullingOn() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + // The SSBO variant without greedy meshing: a greedy-meshed quad takes + // its tile counts from the record's flags. This test exercises the + // ordinary packed-face path, without greedy tiling. + ShaderCorpus.ShaderVariant variant = + ShaderCorpus.Variants().First(v => v.UseSsbo == 1 && v.GreedyMesh == 0); + List stages = + ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkopaque"); + BindEveryDeclaredSampler(device!, seam, programId); + + int target = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + + // Four vertices and six indices, sized as the game sizes a pool: the + // xyz figure is positions, and the device scales it for face records. + int mesh = seam.CreateEmptyMesh( + xyzSize: 4 * 12, normalsSize: 0, uvSize: 0, rgbaSize: 4 * 4, flagsSize: 0, + indicesSize: 6 * sizeof(int), null, null, null, null, + EnumDrawMode.Triangles, staticDraw: false, ssbo: true); + Assert.True(mesh > 0, seam.GetError() ?? "SSBO mesh allocation failed"); + + // In the game's order: the ordinary vertex data first, then the face + // records. The MeshData carries a (zero) xyz array like the game's + // does, which must not reach the storage slot - the device skips it + // for an SSBO mesh, and this is where that is exercised. + var colours = new MeshData(4, 6) { Rgba = new byte[16], RgbaOffset = 0, VerticesCount = 4 }; + Array.Fill(colours.Rgba, (byte)255); + seam.UpdateMesh(mesh, colours); + + // A quad from (-0.5,-0.5) to (0.5,0.5): xyz is the first corner, xyzA + // half of the edge to the second, xyzB half of the edge to the fourth. + var record = CreateFaceRecord(); + record[0] = -0.5f; record[1] = -0.5f; record[2] = 0f; // xyz + record[4] = 0.5f; record[5] = 0f; record[6] = 0f; // xyzA + record[12] = 0f; record[13] = 0.5f; record[14] = 0f; // xyzB + fixed (float* bytes = record) + { + seam.UpdateMeshStorageBuffer(mesh, (IntPtr)bytes, 0, 64); + } + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 1f, 0f, 1f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(true); + seam.SetDepthFunc(0x203); // GL_LEQUAL + // On, as the chunk pass has it. The quad winds counter-clockwise in + // GL's terms, so it is a front face and must survive. + seam.SetCullFace(true); + seam.SetBlend(false, EnumBlendMode.Standard); + + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + seam.Present(); + + int centre = (Size / 2 * Size + Size / 2) * 4; + int corner = (2 * Size + 2) * 4; + byte[] pixels = ReadTarget(seam, framebuffer); + _output.WriteLine($"multi-draw, culling on: centre RGBA = {pixels[centre]}, {pixels[centre + 1]}, " + + $"{pixels[centre + 2]}, {pixels[centre + 3]}"); + bool rasterised = !IsClearColour(pixels, centre); + + // A failure is only useful if it says which part failed, so the same + // face is tried again with culling off and through the single-draw + // path, and the message reports what each of those did. + string diagnosis = ""; + if (!rasterised) + { + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 1f, 0f, 1f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + seam.SetCullFace(false); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + seam.Present(); + bool withoutCulling = !IsClearColour(ReadTarget(seam, framebuffer), centre); + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 1f, 0f, 1f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + seam.SetCullFace(true); + seam.DrawMesh(mesh); + seam.Present(); + bool singleDraw = !IsClearColour(ReadTarget(seam, framebuffer), centre); + + diagnosis = $" (culling off: {(withoutCulling ? "rasterised" : "nothing")}; " + + $"single draw with culling: {(singleDraw ? "rasterised" : "nothing")}; " + + $"diagnostics: {seam.GetError() ?? "none"})"; + } + + Assert.True(rasterised, "the face record did not rasterise through the multi-draw path" + diagnosis); + Assert.True(IsClearColour(pixels, corner), "the face covered the whole target"); + + AssertClean(seam); + } + } + + /// + /// A face record's UV must reach the atlas as the record wrote it. + /// + /// The chunk shaders unpack UVs out of the storage record rather than a + /// vertex attribute: vdata.uv is the origin as 16-bit fixed point and + /// vdata.uvSize the span, and UnpackUv divides both by 32768. Both are + /// ints sitting between the record's vec3s, so any disagreement between the + /// struct C# writes and the one the shader reads lands there first - and + /// misreading the span for the origin, or the other way round, maps a swathe + /// of the atlas across a single block face instead of one block texture. + /// + /// So this draws the same face twice against a four-texel atlas, once with + /// the origin on the red texel and once on the green one, with a zero span + /// both times. Reading back a red face and then a green one is only possible + /// if the record's origin arrived exactly and the span really was zero. + /// + [SkippableFact] + public unsafe void AFaceRecordSamplesTheAtlasWhereItsPackedUvPointsTo() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + ShaderCorpus.ShaderVariant variant = + ShaderCorpus.Variants().First(v => v.UseSsbo == 1 && v.GreedyMesh == 0); + List stages = + ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkopaque"); + int nextUnit = BindEveryDeclaredSampler(device!, seam, programId); + + // A two-by-two atlas: red, green on the bottom row, blue and white on + // the top. Nearest filtering, so a UV inside a texel is that texel + // and nothing is blended in from its neighbours. + var atlasPixels = new byte[] + { + 255, 0, 0, 255, 0, 255, 0, 255, + 0, 0, 255, 255, 255, 255, 255, 255, + }; + int atlas; + fixed (byte* source = atlasPixels) + { + atlas = seam.CreateTexture2D(2, 2, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)source, false); + } + seam.SetTextureParameter(atlas, OptimumGlConstants.TextureMinFilter, 9728); + seam.SetTextureParameter(atlas, OptimumGlConstants.TextureMagFilter, 9728); + + // Both terrain samplers: the base texture and the one the colormap + // include samples through. + foreach (string samplerName in new[] { "terrainTex", "terrainTexLinear" }) + { + seam.SetSamplerUnit(programId, samplerName, nextUnit); + seam.BindTexture(nextUnit, atlas); + nextUnit++; + } + + int target = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + + int mesh = seam.CreateEmptyMesh( + xyzSize: 4 * 12, normalsSize: 0, uvSize: 0, rgbaSize: 4 * 4, flagsSize: 0, + indicesSize: 6 * sizeof(int), null, null, null, null, + EnumDrawMode.Triangles, staticDraw: false, ssbo: true); + Assert.True(mesh > 0, seam.GetError() ?? "SSBO mesh allocation failed"); + + var colours = new MeshData(4, 6) { Rgba = new byte[16], RgbaOffset = 0, VerticesCount = 4 }; + Array.Fill(colours.Rgba, (byte)255); + seam.UpdateMesh(mesh, colours); + + // The record as FaceData writes it: xyz then the packed origin at + // offset 12, the two half-edges, and the packed span at offset 28. + byte[] DrawAt(float u, float v) + { + var record = CreateFaceRecord(); + record[0] = -0.5f; record[1] = -0.5f; record[2] = 0f; // xyz + record[4] = 0.5f; record[5] = 0f; record[6] = 0f; // xyzA + record[12] = 0f; record[13] = 0.5f; record[14] = 0f; // xyzB + + int packedUv = (int)(u * 32768f + 0.5f) + ((int)(v * 32768f + 0.5f) << 16); + var record32 = new int[16]; + Buffer.BlockCopy(record, 0, record32, 0, 64); + record32[3] = packedUv; // uv + record32[7] = 0; // uvSize: no span, so every corner samples the origin + + fixed (int* bytes = record32) + { + seam.UpdateMeshStorageBuffer(mesh, (IntPtr)bytes, 0, 64); + } + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + SetFloat(seam, programId, "subpixelPaddingX", 0f); + SetFloat(seam, programId, "subpixelPaddingY", 0f); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(true); + seam.SetDepthFunc(0x203); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + seam.Present(); + return ReadTarget(seam, framebuffer); + } + + int centre = (Size / 2 * Size + Size / 2) * 4; + + // The centre of the bottom-left texel, then of the bottom-right one. + byte[] onRed = DrawAt(0.25f, 0.25f); + byte[] onGreen = DrawAt(0.75f, 0.25f); + + _output.WriteLine($"origin on red -> {onRed[centre]}, {onRed[centre + 1]}, {onRed[centre + 2]}"); + _output.WriteLine($"origin on green -> {onGreen[centre]}, {onGreen[centre + 1]}, {onGreen[centre + 2]}"); + + // Lighting scales the sampled colour, so the check is which channel + // came through, not how bright it is. Anything drawn at all rules out + // a face that never rasterised. + Assert.True(onRed[centre] > 0 || onGreen[centre + 1] > 0, + "the face did not rasterise, so nothing was sampled"); + Assert.True(onRed[centre] > onRed[centre + 1], + $"a UV origin on the red texel sampled elsewhere: " + + $"{onRed[centre]}, {onRed[centre + 1]}, {onRed[centre + 2]}"); + Assert.True(onGreen[centre + 1] > onGreen[centre], + $"a UV origin on the green texel sampled elsewhere: " + + $"{onGreen[centre]}, {onGreen[centre + 1]}, {onGreen[centre + 2]}"); + + AssertClean(seam); + } + } + + /// + /// Most real chunk faces have a negative UV span, and the shader recovers it + /// by sign extension rather than by reading a signed field. + /// + /// FaceData packs the span as two 15-bit fields and turns a negative delta + /// into its positive complement first - a du of -1/128 is stored as 32512 - + /// so UnpackUv subtracts 32768 back off whenever the field's top bit is set: + /// (uvs & 0x7FFF) - ((uvs & 0x4000) << 1) for u, and the same + /// for v out of the high half. Lose either of those and the span flips from + /// a fraction of a block texture to very nearly the whole atlas, which maps + /// a swathe of unrelated block textures across every face. + /// + /// The packed values here are the ones a real world produced, read back out + /// of the storage buffer at draw time. + /// + [SkippableFact] + public unsafe void AFaceRecordWithANegativeUvSpanStaysOnItsOwnBlockTexture() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + ShaderCorpus.ShaderVariant variant = + ShaderCorpus.Variants().First(v => v.UseSsbo == 1 && v.GreedyMesh == 0); + List stages = + ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkopaque"); + int nextUnit = BindEveryDeclaredSampler(device!, seam, programId); + + // An eight-by-eight atlas that is green everywhere except the one + // texel this face's UVs fall in, which is red. A correct unpack sees + // only that texel; a span that lost its sign sweeps most of the + // atlas and drags the green in. + const int atlasSize = 8; + var atlasPixels = new byte[atlasSize * atlasSize * 4]; + for (int i = 0; i < atlasSize * atlasSize; i++) + { + atlasPixels[i * 4] = 0; + atlasPixels[i * 4 + 1] = 255; + atlasPixels[i * 4 + 2] = 0; + atlasPixels[i * 4 + 3] = 255; + } + + // uv origin (2048, 15488) / 32768 = (0.0625, 0.4727): column 0, row 3. + int redTexel = (3 * atlasSize + 0) * 4; + atlasPixels[redTexel] = 255; + atlasPixels[redTexel + 1] = 0; + + int atlas; + fixed (byte* source = atlasPixels) + { + atlas = seam.CreateTexture2D(atlasSize, atlasSize, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)source, false); + } + seam.SetTextureParameter(atlas, OptimumGlConstants.TextureMinFilter, 9728); + seam.SetTextureParameter(atlas, OptimumGlConstants.TextureMagFilter, 9728); + + foreach (string samplerName in new[] { "terrainTex", "terrainTexLinear" }) + { + seam.SetSamplerUnit(programId, samplerName, nextUnit); + seam.BindTexture(nextUnit, atlas); + nextUnit++; + } + + int target = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + + int mesh = seam.CreateEmptyMesh( + xyzSize: 4 * 12, normalsSize: 0, uvSize: 0, rgbaSize: 4 * 4, flagsSize: 0, + indicesSize: 6 * sizeof(int), null, null, null, null, + EnumDrawMode.Triangles, staticDraw: false, ssbo: true); + Assert.True(mesh > 0, seam.GetError() ?? "SSBO mesh allocation failed"); + + var colours = new MeshData(4, 6) { Rgba = new byte[16], RgbaOffset = 0, VerticesCount = 4 }; + Array.Fill(colours.Rgba, (byte)255); + seam.UpdateMesh(mesh, colours); + + var record = CreateFaceRecord(); + record[0] = -0.5f; record[1] = -0.5f; record[2] = 0f; + record[4] = 0.5f; record[5] = 0f; record[6] = 0f; + record[12] = 0f; record[13] = 0.5f; record[14] = 0f; + + var record32 = new int[16]; + Buffer.BlockCopy(record, 0, record32, 0, 64); + record32[3] = unchecked((int)0x3C800800); // uv: 2048, 15488 + record32[7] = unchecked((int)0x7E007F00); // uvSize: -256, -512 + + fixed (int* bytes = record32) + { + seam.UpdateMeshStorageBuffer(mesh, (IntPtr)bytes, 0, 64); + } + + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 0f, 0f, 1f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + SetFloat(seam, programId, "subpixelPaddingX", 0f); + SetFloat(seam, programId, "subpixelPaddingY", 0f); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(true); + seam.SetDepthFunc(0x203); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + seam.Present(); + + byte[] pixels = ReadTarget(seam, framebuffer); + + // Four points spread across the face: with a span of a fraction of a + // texel every one of them is the red texel, whereas a lost sign puts + // three of the corners somewhere else entirely. + foreach ((int x, int y) in new[] { (Size / 2, Size / 2), (24, 24), (40, 24), (24, 40) }) + { + int at = (y * Size + x) * 4; + _output.WriteLine($"({x},{y}) -> {pixels[at]}, {pixels[at + 1]}, {pixels[at + 2]}"); + Assert.True(pixels[at] > pixels[at + 1], + $"the face sampled off its own block texture at ({x},{y}): " + + $"{pixels[at]}, {pixels[at + 1]}, {pixels[at + 2]}"); + } + + AssertClean(seam); + } + } + + /// + /// Rebinding a sampler's texture between two draws of the same mesh in one + /// frame has to reach the second draw. + /// + /// This is the shape of the chunk pass: the renderer walks the atlas pages, + /// binds page i to terrainTex and terrainTexLinear, draws the pools that + /// belong to that page, and moves on - all within one frame and, for a mesh + /// that spans pages, on the same mesh. A descriptor set cached per program + /// and unit rather than per texture would serve the first page's atlas to + /// every later draw, which puts real block textures on blocks they do not + /// belong to. + /// + [SkippableFact] + public unsafe void RebindingAnAtlasBetweenDrawsChangesWhatTheSecondDrawSamples() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(TryCreateDevice(_output, out VulkanDevice? device), "No usable Vulkan device."); + + using (device) + { + VulkanDevice seam = device!; + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + ShaderCorpus.ShaderVariant variant = + ShaderCorpus.Variants().First(v => v.UseSsbo == 1 && v.GreedyMesh == 0); + List stages = + ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + Assert.NotEmpty(stages); + + int programId = LinkFromCorpus(seam, stages, "chunkopaque"); + int nextUnit = BindEveryDeclaredSampler(device!, seam, programId); + + int SolidPage(byte r, byte g, byte b) + { + var texels = new byte[] { r, g, b, 255 }; + fixed (byte* source = texels) + { + int id = seam.CreateTexture2D(1, 1, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)source, false); + seam.SetTextureParameter(id, OptimumGlConstants.TextureMinFilter, 9728); + seam.SetTextureParameter(id, OptimumGlConstants.TextureMagFilter, 9728); + return id; + } + } + + int firstPage = SolidPage(255, 0, 0); + int secondPage = SolidPage(0, 255, 0); + + int terrainUnit = nextUnit; + int linearUnit = nextUnit + 1; + seam.SetSamplerUnit(programId, "terrainTex", terrainUnit); + seam.SetSamplerUnit(programId, "terrainTexLinear", linearUnit); + + int target = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + + int mesh = seam.CreateEmptyMesh( + xyzSize: 4 * 12, normalsSize: 0, uvSize: 0, rgbaSize: 4 * 4, flagsSize: 0, + indicesSize: 6 * sizeof(int), null, null, null, null, + EnumDrawMode.Triangles, staticDraw: false, ssbo: true); + Assert.True(mesh > 0, seam.GetError() ?? "SSBO mesh allocation failed"); + + var colours = new MeshData(4, 6) { Rgba = new byte[16], RgbaOffset = 0, VerticesCount = 4 }; + Array.Fill(colours.Rgba, (byte)255); + seam.UpdateMesh(mesh, colours); + + var record = CreateFaceRecord(); + record[0] = -0.5f; record[1] = -0.5f; record[2] = 0f; + record[4] = 0.5f; record[5] = 0f; record[6] = 0f; + record[12] = 0f; record[13] = 0.5f; record[14] = 0f; + fixed (float* bytes = record) + { + seam.UpdateMeshStorageBuffer(mesh, (IntPtr)bytes, 0, 64); + } + + // Both pages drawn in one frame, exactly as the chunk pass does it. + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 0f, 0f, 1f, 1f); + seam.ClearDepth(1f); + seam.UseProgram(programId); + SetIdentityMatrices(seam, programId); + SetViewUniforms(seam, programId); + SetFloat(seam, programId, "subpixelPaddingX", 0f); + SetFloat(seam, programId, "subpixelPaddingY", 0f); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + + seam.BindTexture(terrainUnit, firstPage); + seam.BindTexture(linearUnit, firstPage); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + + seam.BindTexture(terrainUnit, secondPage); + seam.BindTexture(linearUnit, secondPage); + seam.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, ssbo: true); + + seam.Present(); + + byte[] pixels = ReadTarget(seam, framebuffer); + int centre = (Size / 2 * Size + Size / 2) * 4; + _output.WriteLine($"after rebinding to the green page -> " + + $"{pixels[centre]}, {pixels[centre + 1]}, {pixels[centre + 2]}"); + + Assert.True(pixels[centre + 1] > pixels[centre], + "the second draw kept sampling the first page's atlas: " + + $"{pixels[centre]}, {pixels[centre + 1]}, {pixels[centre + 2]}"); + + AssertClean(seam); + } + } + + private static unsafe byte[] ReadTarget(VulkanDevice seam, int framebuffer) + { + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.BindFramebuffer(framebuffer); + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + return pixels; + } + + private static bool IsClearColour(byte[] pixels, int offset) => + pixels[offset] >= 250 && pixels[offset + 1] <= 5 && pixels[offset + 2] >= 250; + + private static int LinkFromCorpus( + VulkanDevice seam, List stages, string name) + { + var program = new Program { PassName = name }; + + foreach (ShaderStageSource stage in stages) + { + var shader = new Shader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode ?? "", + }; + Assert.True(seam.CompileShader(shader), name + ": " + (seam.GetError() ?? "compile failed")); + + if (stage.Stage == EnumShaderType.VertexShader) program.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) program.FragmentShader = shader; + else program.GeometryShader = shader; + } + + int programId = seam.LinkProgram(program); + Assert.True(programId > 0, name + ": " + (seam.GetError() ?? "link failed")); + return programId; + } + + /// + /// Binds a one-texel white texture to every sampler the program declares. + /// + /// The device skips a draw whose descriptor set is incomplete, which is the + /// right behaviour but would make this test pass by not drawing at all. The + /// real client binds the atlases; here a stand-in is enough to make the draw + /// legal. + /// + /// The first texture unit the program did not claim. + private static unsafe int BindEveryDeclaredSampler( + VulkanDevice device, VulkanDevice seam, int programId) + { + var white = new byte[] { 255, 255, 255, 255 }; + int unit = 0; + + foreach (string samplerName in device.SamplerNamesOf(programId)) + { + int texture; + fixed (byte* pixels = white) + { + texture = seam.CreateTexture2D(1, 1, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + } + seam.SetSamplerUnit(programId, samplerName, unit); + seam.BindTexture(unit, texture); + unit++; + } + + return unit; + } + + /// + /// The scalar uniforms the chunk passes need in order to draw anything. + /// + /// These are not decoration. chunkopaque computes + /// aTest = outColor.a + ... - lod0Fade and discards when it falls below + /// alphaTest, and lod0Fade is derived from the view distances - left at zero, + /// every fragment in the world fades out and the pass draws nothing at all. + /// The client sets them each frame; a test that does not is testing a + /// configuration the game never runs. + /// + private static void SetViewUniforms(VulkanDevice seam, int programId) + { + SetFloat(seam, programId, "viewDistance", 1024f); + SetFloat(seam, programId, "viewDistanceLod0", 1024f); + SetFloat(seam, programId, "alphaTest", 0.001f); + SetFloat(seam, programId, "zNear", 0.1f); + SetFloat(seam, programId, "zFar", 1024f); + SetFloat(seam, programId, "shadowRangeFar", 1024f); + SetFloat(seam, programId, "shadowRangeNear", 64f); + SetFloat(seam, programId, "shadowMapWidthInv", 1f); + SetFloat(seam, programId, "shadowMapHeightInv", 1f); + int ambient = seam.GetUniformLocation(programId, "rgbaAmbientIn"); + if (ambient >= 0) seam.SetUniform(programId, ambient, 1f, 1f, 1f); + // The underwater include samples at gl_FragCoord / frameSize even in + // an air scene. Leaving this at zero gives undefined texture reads. + int frameSize = seam.GetUniformLocation(programId, "frameSize"); + if (frameSize >= 0) seam.SetUniform(programId, frameSize, (float)Size, (float)Size); + } + + private static void SetFloat(VulkanDevice seam, int programId, string name, float value) + { + int location = seam.GetUniformLocation(programId, name); + if (location >= 0) seam.SetUniform(programId, location, value); + } + + /// + /// The matrices every chunk program multiplies by. Identity leaves the mesh's + /// clip-space positions alone, which is what makes the output checkable. + /// + private static void SetIdentityMatrices(VulkanDevice seam, int programId) + { + float[] identity = + { + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + + foreach (string name in new[] + { + "projectionMatrix", "modelViewMatrix", "modelMatrix", "mvpMatrix", + "toShadowMapSpaceMatrixFar", "toShadowMapSpaceMatrixNear", + }) + { + int location = seam.GetUniformLocation(programId, name); + if (location >= 0) seam.SetUniformMatrix(programId, location, identity); + } + } + + private static void AssertClean(VulkanDevice seam) => GpuTest.AssertClean(seam); +} diff --git a/Optimum.Render.Vulkan.Tests/DeviceValidationTests.cs b/Optimum.Render.Vulkan.Tests/DeviceValidationTests.cs new file mode 100644 index 00000000..36905885 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/DeviceValidationTests.cs @@ -0,0 +1,183 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class DeviceValidationTests +{ + private readonly ITestOutputHelper _output; + public DeviceValidationTests(ITestOutputHelper output) => _output = output; + + private VulkanContext Open(List messages) + { + var options = GpuTest.ContextOptions(messages); + options.ValidationFeatures = "sync,best"; + bool created = VulkanContext.TryCreate(options, out VulkanContext? context, out string? reason); + // An explicitly requested device must fail visibly if it cannot be used. + if (Environment.GetEnvironmentVariable(GpuTest.DeviceIndexVariable) != null) + Assert.True(created, reason); + Skip.IfNot(created, "No usable Vulkan device: " + reason); + try + { + _output.WriteLine($"device={context!.Capabilities.DeviceName}; vendor={context.Capabilities.VendorId:X}; driver={context.Capabilities.DriverVersion}; API={context.Capabilities.ApiVersion}"); + _output.WriteLine("validation=" + context.ValidationSettingsApplied); + Assert.True(context.ValidationEnabled, "Khronos validation must be installed for GPU acceptance."); + Assert.DoesNotContain("NOT APPLIED", context.ValidationSettingsApplied); + Assert.False(string.IsNullOrWhiteSpace(context.ValidationSettingsApplied)); + string? expected = Environment.GetEnvironmentVariable("OPTIMUM_TEST_DEVICE_NAME"); + if (!string.IsNullOrWhiteSpace(expected)) + Assert.Contains(expected, context.Capabilities.DeviceName, StringComparison.OrdinalIgnoreCase); + return context; + } + catch { context!.Dispose(); throw; } + } + + [Fact] + public void ColorWriteOverridesNeverRequireUnsupportedFeatures() + { + foreach (bool enable in new[] { false, true }) + foreach (bool mask in new[] { false, true }) + foreach (string forced in new[] { "enable", "mask", "pipeline" }) + { + var requested = DeviceCaps.ParseColorWriteTier(forced); + string actual = DeviceCaps.Token(DeviceCaps.SelectColorWriteTier(enable, mask, requested)); + string expected = forced == "pipeline" ? "pipeline" + : forced == "enable" && enable ? "enable" : mask ? "mask" : "pipeline"; + Assert.Equal(expected, actual); + } + Assert.Null(DeviceCaps.ParseColorWriteTier("unknown")); + Assert.Equal(DeviceCaps.ParseColorWriteTier("mask"), DeviceCaps.ParseColorWriteTier(" MASK ")); + } + + [SkippableFact] + public void SelectedDeviceCreatesTheActualSharedPipelineLayout() + { + var messages = new List(); + using (var context = Open(messages)) + { + Assert.Empty(DescriptorIndexingFloor.Missing(context.Capabilities.DescriptorIndexing)); + using var layout = SharedPipelineLayout.CreateStandalone(context); + Assert.NotEqual(0UL, layout.Layout.Handle); + Assert.NotEqual(0UL, layout.TextureSetLayout.Handle); + if (context.CheckpointsAvailable) Assert.NotNull(context.ReadQueueCheckpoints()); + } + // Include destruction in the validation boundary. + ValidationAssert.NoErrors(messages); + } + + [SkippableFact] + public void TranslatedFullscreenDrawUsesGlFramebufferCoordinates() + { + var device = GpuTest.CreateDevice(_output); + try + { + var caps = device.ContextForTests.Capabilities; + Assert.True(caps.ApiVersion >= VulkanContext.MinimumApiVersion); + Assert.True(caps.MultiDrawIndirect); + Assert.True(caps.MaxBoundDescriptorSets >= 3); + Assert.True(caps.MaxSamplerLodBias >= 2f); + Assert.True(caps.MaxImageDimension2D >= 4096); + int program = GpuTest.LinkProgram(device, """ + #version 330 core + out vec2 uv; + void main() { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + uv = vec2((x + 1.0) * 0.5, (y + 1.0) * 0.5); + } + """, """ + #version 330 core + in vec2 uv; + out vec4 color; + void main() { color = vec4(uv, 0, 1); } + """, "coordinate-convention"); + const int size = 32; + int image = device.CreateTexture2DRaw(size, size, 0x8058, IntPtr.Zero, 4); + int target = device.CreateFramebuffer(size, size); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + device.BeginFrame(); device.BindFramebuffer(target); device.UseProgram(program); + device.SetViewport(0, 0, size, size); device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.DrawFullscreenTriangle(); + byte[] pixels = device.ReadBackLevel0ForTests(image); + Assert.Equal(size * size * 4, pixels.Length); + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + { + int i = (y * size + x) * 4; + int expectedX = (int)Math.Round((x + 0.5) * 255.0 / size); + int expectedY = (int)Math.Round((y + 0.5) * 255.0 / size); + Assert.InRange((int)pixels[i], expectedX - 1, expectedX + 1); + Assert.InRange((int)pixels[i + 1], expectedY - 1, expectedY + 1); + Assert.Equal(0, pixels[i + 2]); Assert.Equal(255, pixels[i + 3]); + } + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void SynchronizationValidationDetectsAnActualMissingImageBarrier() + { + var messages = new List(); + using (var context = Open(messages)) + using (var commands = new SetupQueue(context)) + using (var textures = new TextureManager(context, commands.Uploads)) + { + var source = textures.Get(textures.Create(4, 4, Format.R8G8B8A8Unorm))!; + var target = textures.Get(textures.Create(4, 4, Format.R8G8B8A8Unorm))!; + commands.SubmitAndWait(commandBuffer => + { + textures.TransitionTexture(commandBuffer, source, ImageLayout.TransferSrcOptimal); + textures.TransitionTexture(commandBuffer, target, ImageLayout.TransferDstOptimal); + var region = new ImageCopy + { + SrcSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + DstSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + Extent = new Extent3D(4, 4, 1), + }; + for (int i = 0; i < 2; i++) + context.Api.CmdCopyImage(commandBuffer, source.Image, ImageLayout.TransferSrcOptimal, + target.Image, ImageLayout.TransferDstOptimal, 1, ®ion); + }); + } + var snapshot = ValidationAssert.Snapshot(messages); + foreach (string message in snapshot) _output.WriteLine(message); + Assert.Contains(snapshot, m => m.Contains("SYNC-HAZARD-WRITE-AFTER-WRITE", StringComparison.Ordinal)); + // This one deliberate hazard is the control, not an allowance for renderer errors. + ValidationAssert.NoErrors(snapshot.Where(m => !m.Contains("SYNC-HAZARD-WRITE-AFTER-WRITE", StringComparison.Ordinal)).ToArray()); + } +} + +internal static class ValidationAssert +{ + public static List Snapshot(IReadOnlyCollection messages) + { + lock (messages) return new List(messages); + } + + public static bool IsSynchronization(string message) => + message.Contains("[SYNC-", StringComparison.Ordinal); + + public static void NoErrors(IReadOnlyCollection messages) + { + string[] errors = Snapshot(messages).Where(m => + m.StartsWith(VulkanContext.ErrorPrefix, StringComparison.Ordinal) || IsSynchronization(m)).ToArray(); + Assert.True(errors.Length == 0, "validation errors:\n" + string.Join("\n", errors)); + } + + // Retained for existing call sites while the feature suite is replaced. + public static void NoSyncHazards(IReadOnlyCollection messages, string callerFile = "") + { + string[] hazards = Snapshot(messages).Where(IsSynchronization).ToArray(); + Assert.True(hazards.Length == 0, "synchronization hazards:\n" + string.Join("\n", hazards)); + } +} diff --git a/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/ModPassFixtureSystem.cs b/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/ModPassFixtureSystem.cs new file mode 100644 index 00000000..ff84ce54 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/ModPassFixtureSystem.cs @@ -0,0 +1,118 @@ +// Source: Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/ModPassFixtureSystem.cs +namespace Optimum.Render.Vulkan.Tests.Fixtures.ModPassFixture +{ +using System; +using Vintagestory.API.Client; +using Vintagestory.API.Common; + +/// +/// A fixture mod that follows docs/vulkan.md step by step: one declared pass and one +/// motion writer, registered in StartClientSide through the client-API extensions and removed in +/// Dispose. The pass tints Primary's colour from its glow after the AfterOIT renderers and writes +/// motion for its draw, so it reads one attachment of its own target (which therefore leaves the +/// rendering scope), writes colour and depth, and opens the motion window. The writer is what a +/// RegisterRenderer renderer of the same mod would open around its own draws. +/// +/// What the draw does is supplied by the host (): in a shipped mod it would draw +/// through capi.Render with its own shader; the GPU test draws the same shape through the platform. +/// +public sealed class ModPassFixtureSystem : ModSystem +{ + public const string PassName = "glow-tint"; + public const string WriterName = "fixture-renderer"; + + public OptimumPassDecl? Pass { get; private set; } + + public OptimumMotionWriterDecl? Writer { get; private set; } + + /// The draw the pass performs; null draws nothing. + public Action? Drawer; + + /// The id registrations are stored under (the mod id, or this assembly's name when loaded outside the mod loader). + public string ModId => OptimumModRenderExtensions.OptimumModId(this); + + public override bool ShouldLoad(EnumAppSide forSide) => forSide == EnumAppSide.Client; + + public override void StartClientSide(ICoreClientAPI api) + { + Writer = new OptimumMotionWriterDecl { Name = WriterName, Mode = EnumOptimumMotionWrite.WithColor }; + if (!api.RegisterOptimumMotionWriter(this, Writer, out string reason)) + throw new InvalidOperationException(reason); + + Pass = new OptimumPassDecl + { + Name = PassName, + Slot = EnumOptimumPass.AfterOIT, + Reads = new[] { EnumOptimumAttachment.PrimaryGlow }, + Writes = new[] { EnumOptimumAttachment.PrimaryColor, EnumOptimumAttachment.PrimaryDepth }, + Draw = OnDraw, + MotionWriter = new OptimumMotionWriterDecl { Name = PassName, Mode = EnumOptimumMotionWrite.WithColor }, + }; + if (!api.RegisterOptimumPass(this, Pass, out reason)) + throw new InvalidOperationException(reason); + } + + private void OnDraw(OptimumPassDecl pass) => Drawer?.Invoke(pass); + + public override void Dispose() + { + OptimumModPasses.UnregisterMod(ModId); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/Fixtures/ClientApiStub.cs +namespace Optimum.Render.Vulkan.Tests.Fixtures +{ +using System; +using System.Collections.Generic; +using System.Reflection; +using Vintagestory.API.Client; + +/// +/// A client API with only the members mod registration touches: Event and its +/// LeaveWorld event. Everything else returns its default. +/// +public class ClientApiStub : DispatchProxy +{ + public readonly List LeaveWorldHandlers = new(); + + private IClientEventAPI? events; + + public static (ICoreClientAPI Api, ClientApiStub Stub) Create() + { + ICoreClientAPI api = Create(); + var stub = (ClientApiStub)(object)api; + IClientEventAPI events = Create(); + var eventStub = (ClientApiStub)(object)events; + stub.events = events; + eventStub.owner = stub; + return (api, stub); + } + + private ClientApiStub? owner; + + /// What leaving the world does to the subscribers. + public void LeaveWorld() + { + foreach (Action handler in LeaveWorldHandlers.ToArray()) handler(); + } + + protected override object? Invoke(MethodInfo? targetMethod, object?[]? args) + { + switch (targetMethod!.Name) + { + case "get_Event": + return events; + case "add_LeaveWorld": + (owner ?? this).LeaveWorldHandlers.Add((Action)args![0]!); + return null; + case "remove_LeaveWorld": + (owner ?? this).LeaveWorldHandlers.Remove((Action)args![0]!); + return null; + } + Type type = targetMethod.ReturnType; + return type.IsValueType && type != typeof(void) ? Activator.CreateInstance(type) : null; + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/modinfo.json b/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/modinfo.json new file mode 100644 index 00000000..2b7e69ff --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Fixtures/ModPassFixture/modinfo.json @@ -0,0 +1,9 @@ +{ + "type": "code", + "modid": "optimummodpassfixture", + "name": "Optimum mod pass fixture", + "description": "Declares one Vulkan frame-graph pass and one motion writer, following docs/vulkan.md.", + "version": "1.0.0", + "side": "Client", + "dependencies": { "game": "" } +} diff --git a/Optimum.Render.Vulkan.Tests/FramePlanningTests.cs b/Optimum.Render.Vulkan.Tests/FramePlanningTests.cs new file mode 100644 index 00000000..5de9e8df --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/FramePlanningTests.cs @@ -0,0 +1,190 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Xunit; + +namespace Optimum.Render.Vulkan.Tests; + +public class FramePlanningTests +{ + private static PassSignature Pass(int target, bool transient = true, params int[] reads) => new() + { + NameId = target, Width = 32, Height = 24, FormatsId = 1, + Attachments = new[] { new AttachmentUse(target, ResourceUsage.ColorWrite, transient) }, + Reads = reads, + }; + + [Fact] + public void PostChainDiscardsOnlyTransientInitialContents() + { + var frame = new[] { Pass(1), Pass(2, true, 1), Pass(3, true, 2), Pass(4, false, 3) }; + var plan = FramePlan.Build(frame); + for (int i = 0; i < 3; i++) + { + Assert.Equal(AttachmentLoadOp.DontCare, plan.LoadOp(i, 0)); + } + Assert.Equal(AttachmentLoadOp.Load, plan.LoadOp(3, 0)); + } + + [Theory] + [InlineData(ResourceUsage.ColorBlend)] + [InlineData(ResourceUsage.DepthReadOnly)] + public void InitialReadsPreserveExistingContents(ResourceUsage firstUse) + { + var pass = Pass(1); + pass.Attachments[0] = new AttachmentUse(1, firstUse, true); + var plan = FramePlan.Build(new[] { pass }); + Assert.Equal(AttachmentLoadOp.Load, plan.LoadOp(0, 0)); + } + + [Fact] + public void CachedPlansOwnTheirDataAndMismatchesFallBackToLoading() + { + var frame = new[] { Pass(1), Pass(2, true, 1) }; + var original = frame.Select(p => p.Clone()).ToArray(); + var plan = FramePlan.Build(frame); + frame[0].Attachments[0] = new AttachmentUse(99, ResourceUsage.ColorBlend, false); + frame[1].Reads[0] = 99; + Assert.True(plan.Matches(original)); + Assert.False(plan.Matches(frame)); + var graph = new FrameGraph { Enabled = true }; + foreach (var pass in original) graph.OpenPass(pass, true); + graph.EndFrame(); + foreach (var pass in frame) + { + int index = graph.OpenPass(pass, true); + Assert.Equal(AttachmentLoadOp.Load, graph.PlannedLoad(index, 0)); + } + } + + [Fact] + public void AlternatingHistoryShapesReusePlansUntilTheirInputsChange() + { + var graph = new FrameGraph { Enabled = true }; + var a = Pass(10, true, 11); + var b = Pass(11, true, 10); + graph.OpenPass(a, true); graph.EndFrame(); + var planA = graph.Plan; + graph.OpenPass(b, true); graph.EndFrame(); + var planB = graph.Plan; + for (int i = 0; i < 6; i++) + { + var pass = i % 2 == 0 ? a : b; + int index = graph.OpenPass(pass, true); + Assert.Equal(AttachmentLoadOp.DontCare, graph.PlannedLoad(index, 0)); + graph.EndFrame(); + Assert.Same(i % 2 == 0 ? planA : planB, graph.Plan); + } + var resized = a.Clone(); resized.Height++; + int changed = graph.OpenPass(resized, true); + Assert.Equal(AttachmentLoadOp.Load, graph.PlannedLoad(changed, 0)); + graph.EndFrame(); + Assert.NotSame(planA, graph.Plan); + } + + [Fact] + public void AllocatorPreservesInclusiveLifetimesAndDistinctImageDescriptions() + { + var random = new Random(901); + var intervals = Enumerable.Range(0, 96).Select(i => + { + int first = random.Next(20); + return (Description: new TransientImageDesc(32, 24, Format.R8G8B8A8Unorm, (uint)(i % 3 + 1)), FirstPass: first, LastPass: first + random.Next(5)); + }).OrderBy(i => i.FirstPass).ToArray(); + var allocator = new TransientAllocator(new ImageBacking(), aliasing: true); + allocator.BeginFrame(); + int[] slots = intervals.Select(i => allocator.Acquire(i.Description, i.FirstPass, i.LastPass).Slot).ToArray(); + for (int i = 0; i < intervals.Length; i++) + for (int j = i + 1; j < intervals.Length; j++) + if (slots[i] == slots[j]) + { + Assert.Equal(intervals[i].Description, intervals[j].Description); + Assert.True(intervals[i].LastPass < intervals[j].FirstPass || intervals[j].LastPass < intervals[i].FirstPass); + } + int minimum = intervals.GroupBy(i => i.Description).Sum(group => + Enumerable.Range(0, 25).Max(pass => group.Count(i => i.FirstPass <= pass && pass <= i.LastPass))); + Assert.Equal(minimum, slots.Distinct().Count()); + } + + private sealed class ImageBacking : ITransientBacking + { + private int _next; + public int Create(TransientImageDesc desc) => ++_next; + public void Destroy(int textureId) { } + public ulong BytesOf(int textureId) => 0; + public bool TryDescribe(int textureId, out TransientImageDesc desc) { desc = default; return false; } + public void Discard(int textureId) { } + public void Rebind(int logicalTextureId, int physicalTextureId) { } + public void RestoreBindings() { } + } + + [Theory] + [InlineData(ResourceUsage.StorageReadCompute)] + [InlineData(ResourceUsage.StorageWrite)] + [InlineData(ResourceUsage.StorageReadWrite)] + public void ConsecutiveComputeUsesOrderPreviousWritesEvenInTheSameStage(ResourceUsage next) + { + var state = new ResourceStateTracker(1, 1, false); + var transitions = new List(); + state.Require(0, 1, 0, 1, ResourceUsage.StorageWrite, false, transitions); + transitions.Clear(); + state.Require(0, 1, 0, 1, next, false, transitions); + var barrier = Assert.Single(transitions).Sides; + Assert.Equal(ImageLayout.General, barrier.OldLayout); + Assert.Equal(ImageLayout.General, barrier.NewLayout); + Assert.True((barrier.SrcAccess & AccessFlags2.ShaderStorageWriteBit) != 0); + Assert.True((barrier.DstStage & PipelineStageFlags2.ComputeShaderBit) != 0); + } + + [Fact] + public void ASecondReaderStageMustAlsoObserveTheProducer() + { + var state = new ResourceStateTracker(1, 1, false); + var transitions = new List(); + state.Require(0, 1, 0, 1, ResourceUsage.TransferDst, false, transitions); + state.Require(0, 1, 0, 1, ResourceUsage.SampleVertex, false, transitions); + transitions.Clear(); + state.Require(0, 1, 0, 1, ResourceUsage.SampleFragment, false, transitions); + var barrier = Assert.Single(transitions).Sides; + Assert.True((barrier.SrcAccess & AccessFlags2.TransferWriteBit) != 0); + Assert.True((barrier.DstStage & PipelineStageFlags2.FragmentShaderBit) != 0); + transitions.Clear(); + state.Require(0, 1, 0, 1, ResourceUsage.SampleFragment, false, transitions); + Assert.Empty(transitions); + } + + [Fact] + public void MipTransitionsDoNotChangeUntouchedLevelsAndDiscardStillOrdersPriorUses() + { + var state = new ResourceStateTracker(3, 1, false); + var transitions = new List(); + state.Require(0, 3, 0, 1, ResourceUsage.TransferDst, false, transitions); + state.Require(1, 1, 0, 1, ResourceUsage.SampleFragment, false, transitions); + Assert.Equal(ImageLayout.TransferDstOptimal, state.StateOf(0, 0).Layout); + Assert.Equal(ImageLayout.ShaderReadOnlyOptimal, state.StateOf(1, 0).Layout); + Assert.Equal(ImageLayout.TransferDstOptimal, state.StateOf(2, 0).Layout); + state.Discard(); transitions.Clear(); + state.Require(1, 1, 0, 1, ResourceUsage.ColorWrite, false, transitions); + var barrier = Assert.Single(transitions).Sides; + Assert.Equal(ImageLayout.Undefined, barrier.OldLayout); + Assert.True((barrier.SrcStage & PipelineStageFlags2.FragmentShaderBit) != 0); + Assert.Equal(ImageLayout.ColorAttachmentOptimal, barrier.NewLayout); + } + + [Fact] + public void ComputeMipBindingsCoverOddExtentsAndRejectFeedbackOnTheSameLevel() + { + var pass = new ComputePassDeclaration + { + Bindings = new[] { new ComputeBinding(0, 1, ComputeAccess.Sampled), new ComputeBinding(1, 1, ComputeAccess.StorageWrite, 1) }, + Dispatches = new[] { ComputeDispatch.Covering(1) }, + }; + ComputeImageInfo? Image(int _) => new ComputeImageInfo(35, 19, 4, 1); + Assert.Null(ComputePassPlanner.Validate(pass, Image)); + Assert.Equal((3u, 2u, 1u), ComputePassPlanner.Groups(pass.Dispatches[0], pass, Image, 8, 8)); + pass.Bindings[1] = pass.Bindings[1] with { BaseMip = 0 }; + Assert.NotNull(ComputePassPlanner.Validate(pass, Image)); + } +} diff --git a/Optimum.Render.Vulkan.Tests/GlShapedDevice.cs b/Optimum.Render.Vulkan.Tests/GlShapedDevice.cs new file mode 100644 index 00000000..f6a6dcd4 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/GlShapedDevice.cs @@ -0,0 +1,277 @@ +using System; +using System.Collections.Generic; +using System.Runtime.CompilerServices; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Optimum.Render.Vulkan.Platform; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The OpenGL-shaped calls the GPU tests are written in, on a bare : +/// set state, bind, clear, draw, read back. The device has no GL state machine any more; these +/// record what a test states into a per device, exactly as +/// records the client's statements, and every draw goes through +/// , the platform's generic native draw. A test therefore renders through +/// the route the client's own unrecognised draws take. +/// +/// A device a owns shares the platform's record, program and +/// target (VulkanDevice.OwnerPlatform): a fixture that states on the device and then calls one of +/// the platform's bodies draws with that state, and the device's draws are the platform's. +/// +/// Semantics kept from the removed emulation: a bind names the target the next clears, draws and +/// readbacks address (nothing is drawn before one); a declared pass (DeclarePass) gives the +/// following draws its name, slots, reads and flags until the next declaration, bind of another +/// target or EndPass; a clear honours the stated draw buffers and colour mask, and a depth +/// clear the depth mask. +/// +internal static class GlShapedDevice +{ + private sealed class Record + { + public StatedRenderState Stated = new(); + public VulkanClientPlatform? Platform; + private int program; + + /// glUseProgram: the platform's record when a platform owns the device. + public int Program + { + get => Platform?.statedProgram ?? program; + set + { + program = value; + if (Platform != null) Platform.statedProgram = value; + } + } + + public int Bound; + public PassDeclaration? Declared; + public string? LastRefusal; + public long Refusals; + + } + + private static readonly ConditionalWeakTable Records = new(); + private static Record Of(VulkanDevice device) + { + if (!Records.TryGetValue(device, out Record? record)) + { + record = new Record(); + Records.Add(device, record); + Record captured = record; + device.FramebufferDeleted += id => + { + captured.Stated.ForgetFramebuffer(id); + if (captured.Bound == id) captured.Bound = 0; + }; + } + if (record.Platform == null) + { + VulkanClientPlatform? owner = PlatformOf(device); + if (owner != null) + { + record.Stated = owner.stated; + record.Platform = owner; + } + } + return record; + } + + /// The platform that owns the device, if any. + private static VulkanClientPlatform? PlatformOf(VulkanDevice device) => device.OwnerPlatform; + + /// The default target's id as the stated record keys it. + private static int Key(VulkanDevice device, int framebufferId) => + framebufferId != 0 && framebufferId == device.DefaultFramebufferId ? PassDeclaration.DefaultFramebuffer : framebufferId; + + extension(VulkanDevice device) + { + /// The state the tests stated on this device. + internal StatedRenderState StatedForTests => Of(device).Stated; + + /// Draws the stated route refused, and the last reason. + internal long StatedRefusalsForTests => Of(device).Refusals; + + internal string? LastStatedRefusalForTests => Of(device).LastRefusal; + + // ------------------------------------------------------------ fixed function + + public void UseProgram(int programId) => Of(device).Program = programId; + + public void SetViewport(int x, int y, int width, int height) => + Of(device).Stated.Viewport = new Rect2D(new Offset2D(x, y), + new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + + /// glScissor, clipped to the positive quadrant as the platform clips it. + public void SetScissor(int x, int y, int width, int height) + { + int clippedX = Math.Max(0, x); + int clippedY = Math.Max(0, y); + width -= clippedX - x; + height -= clippedY - y; + Of(device).Stated.Scissor = new Rect2D(new Offset2D(clippedX, clippedY), + new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + } + + public void SetScissorEnabled(bool enabled) => Of(device).Stated.ScissorEnabled = enabled; + + public bool ScissorEnabled => Of(device).Stated.ScissorEnabled; + + public void SetDepthTest(bool enabled) => Of(device).Stated.DepthTest = enabled; + + public void SetDepthMask(bool enabled) => Of(device).Stated.DepthWrite = enabled; + + public void SetDepthFunc(int glFunc) => Of(device).Stated.DepthCompare = GlEnums.CompareOpFrom(glFunc); + + public void SetCullFace(bool enabled) => Of(device).Stated.CullEnabled = enabled; + + public void SetCullFaceMode(bool back) => Of(device).Stated.CullBack = back; + + public void SetBlend(bool enabled, EnumBlendMode mode) + { + StatedRenderState stated = Of(device).Stated; + stated.SetBlendEnabled(enabled); + stated.SetBlendMode(mode); + } + + public void SetBlendEnabled(bool enabled) => Of(device).Stated.SetBlendEnabled(enabled); + + public void SetBlendFuncSeparate(int attachment, int srcColor, int dstColor, int srcAlpha, int dstAlpha) => + Of(device).Stated.SetSlotFunc(attachment, srcColor, dstColor, srcAlpha, dstAlpha); + + public void SetBlendEquation(int attachment, int equation) => + Of(device).Stated.SetSlotEquation(attachment, equation); + + public void SetColorMask(bool r, bool g, bool b, bool a) => Of(device).Stated.SetColorMask(r, g, b, a); + + public void SetStencilTest(bool enabled) => Of(device).Stated.StencilTest = enabled; + + public void SetWireframe(bool enabled) => Of(device).Stated.Wireframe = enabled; + + public void SetLineWidth(float width) => Of(device).Stated.LineWidth = width; + + // ------------------------------------------------------------ textures + + public void BindTexture(int unit, int textureId) => Of(device).Stated.BindTexture(unit, textureId); + + public void BindTextureCube(int unit, int textureId) => Of(device).Stated.BindTexture(unit, textureId); + + public void BindSampler(int unit, int samplerId) => Of(device).Stated.BindSampler(unit, samplerId); + + // ------------------------------------------------------------ targets + + public void SetDrawBuffers(int framebufferId, int attachmentMask) => + Of(device).Stated.SetDrawBuffers(Key(device, framebufferId), (uint)attachmentMask); + + public void BindFramebuffer(int framebufferId) + { + Record record = Of(device); + int key = Key(device, framebufferId); + if (record.Declared != null && record.Declared.FramebufferId != key) SetDeclared(record, null); + record.Bound = key; + if (record.Platform == null) return; + // A raw bind is what a fork renderer does on the platform: its draws address it too. + if (key > 0) record.Platform.NoteForkFramebuffer(key); + else record.Platform.CurrentFrameBuffer = null!; + } + + public void BindDefaultFramebuffer() => device.BindFramebuffer(PassDeclaration.DefaultFramebuffer); + + /// The following draws belong to this pass (0: the bound target, -1: the default one). + internal void DeclarePass(PassDeclaration declaration) + { + Record record = Of(device); + int key = declaration.FramebufferId == PassDeclaration.BoundFramebuffer + ? Target(record) + : Key(device, declaration.FramebufferId); + device.BindFramebuffer(key); + SetDeclared(record, new PassDeclaration + { + Name = declaration.Name, + FramebufferId = key, + ColorSlots = declaration.ColorSlots, + Reads = declaration.Reads, + TransientSlots = declaration.TransientSlots, + Flags = declaration.Flags, + }); + } + + /// Ends the declared pass and closes its scope. + internal void EndPass() + { + SetDeclared(Of(device), null); + device.EndNativePass(); + device.EndStagePass(); + } + + public void ClearColor(int attachment, float r, float g, float b, float a) + { + Record record = Of(device); + int target = Target(record); + if (target == 0) return; + if (((record.Stated.DrawBuffers(target) >> attachment) & 1) == 0 || record.Stated.ColorMask == 0) return; + device.ClearNativeColor(target, attachment, r, g, b, a); + } + + public void ClearDepth(float depth) + { + Record record = Of(device); + int target = Target(record); + if (target == 0 || !record.Stated.DepthWrite) return; + device.ClearNativeDepth(target, depth); + } + + public void ClearStencil() { } + + /// Colour attachment 0 of the bound target, in its own channel order. + public void ReadDefaultFramebuffer(int x, int y, int width, int height, IntPtr destination) + { + int target = Target(Of(device)); + if (target == 0) return; + device.ReadFramebufferColor(target, x, y, width, height, destination); + } + + // ------------------------------------------------------------ draws + + public void DrawMesh(int meshId) => Draw(device, meshId, 1, null, null, 0); + + public void DrawMeshInstanced(int meshId, int instanceCount) + { + if (instanceCount > 0) Draw(device, meshId, instanceCount, null, null, 0); + } + + public void DrawMeshMulti(int meshId, int[] indicesStarts, int[] indicesSizes, int groupCount, bool ssbo) => + Draw(device, meshId, 1, indicesStarts, indicesSizes, groupCount); + + public void DrawFullscreenTriangle() => Draw(device, 0, 1, null, null, 0); + } + + /// The target a clear, draw or readback addresses: the platform's current one when a platform owns the device. + private static int Target(Record record) => record.Platform?.CurrentTargetId ?? record.Bound; + + /// The declared pass, held where the draws read it. + private static void SetDeclared(Record record, PassDeclaration? declared) + { + record.Declared = declared; + if (record.Platform != null) record.Platform.statedPass = declared; + } + + private static void Draw(VulkanDevice device, int meshId, int instances, int[]? starts, int[]? sizes, int groupCount) + { + Record record = Of(device); + if (record.Platform != null) + { + record.Platform.RecordStatedDraw(meshId, instances, starts, sizes, groupCount); + return; + } + if (record.Bound == 0 || record.Program <= 0) return; + if (!StatedDraw.Record(device, record.Stated, record.Program, record.Bound, meshId, instances, + starts, sizes, groupCount, out string? refusal, record.Declared) && refusal != null) + { + record.Refusals++; + record.LastRefusal = refusal; + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/GpuTest.cs b/Optimum.Render.Vulkan.Tests/GpuTest.cs new file mode 100644 index 00000000..1586227c --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/GpuTest.cs @@ -0,0 +1,352 @@ +// Source: Optimum.Render.Vulkan.Tests/GpuTest.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Runtime.CompilerServices; +using Optimum.Render.Vulkan.Core; +using Vintagestory.API.Config; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +/// +/// The one place GPU tests get a or a +/// from. +/// +/// Every context and device comes up with the validation layers and, by +/// default, synchronization validation plus best practices ("sync,best"). +/// Plain validation missed the R32F-history and masked-clear bugs that only +/// synchronization validation names (TAA P2 and P4, 2026-09-10/11), so the +/// suite runs with it unless OPTIMUM_TEST_VALIDATION_FEATURES says otherwise; +/// an empty value turns the extra features off. +/// +internal static class GpuTest +{ + public const string ValidationFeaturesVariable = "OPTIMUM_TEST_VALIDATION_FEATURES"; + public const string DefaultValidationFeatures = "sync,best"; + public const string DeviceIndexVariable = "OPTIMUM_TEST_DEVICE_INDEX"; + + public static string ValidationFeatures => + Environment.GetEnvironmentVariable(ValidationFeaturesVariable) ?? DefaultValidationFeatures; + + private static int DeviceIndex + { + get + { + string? value = Environment.GetEnvironmentVariable(DeviceIndexVariable); + if (value == null) return -1; + Assert.True(int.TryParse(value, out int index) && index >= 0, + DeviceIndexVariable + " must be a nonnegative device index."); + return index; + } + } + + private static void CheckContext(VulkanContext context, ITestOutputHelper output) + { + output.WriteLine($"device={context.Capabilities.DeviceName}; vendor={context.Capabilities.VendorId:X}; driver={context.Capabilities.DriverVersion}; validation={context.ValidationSettingsApplied}"); + Assert.True(context.ValidationEnabled, "Khronos validation must be installed for GPU acceptance."); + Assert.DoesNotContain("NOT APPLIED", context.ValidationSettingsApplied); + if (!string.IsNullOrWhiteSpace(ValidationFeatures)) + Assert.False(string.IsNullOrWhiteSpace(context.ValidationSettingsApplied)); + string? expected = Environment.GetEnvironmentVariable("OPTIMUM_TEST_DEVICE_NAME"); + if (!string.IsNullOrWhiteSpace(expected)) + Assert.Contains(expected, context.Capabilities.DeviceName, StringComparison.OrdinalIgnoreCase); + } + + private static void CheckUnavailable(string? reason, ITestOutputHelper output) + { + output.WriteLine("Vulkan unavailable: " + reason); + Assert.True(Environment.GetEnvironmentVariable(DeviceIndexVariable) == null + && Environment.GetEnvironmentVariable("OPTIMUM_TEST_DEVICE_NAME") == null, + "The explicitly requested GPU could not initialize: " + reason); + } + + public static VulkanContext CreateContext(ITestOutputHelper output, List? messages = null) + { + Skip.IfNot(TryCreateContext(output, messages, out var context), "No usable Vulkan device."); + return context!; + } + + public static VulkanDevice CreateDevice(ITestOutputHelper output, Action? configure = null) + { + Skip.IfNot(TryCreateDevice(output, out var device, configure), "No usable Vulkan device."); + return device!; + } + + /// Headless, validated options; receives every layer message. + public static VulkanContextOptions ContextOptions(List? messages = null) => new() + { + Headless = true, + EnableValidation = true, + ValidationFeatures = ValidationFeatures, + PreferredDeviceIndex = DeviceIndex, + DebugCallback = messages == null ? null : Recorder(messages), + }; + + /// + /// Appends under the list's own lock: the layers call back from whichever + /// thread made the Vulkan call, and the asserts snapshot under the same lock. + /// + public static Action Recorder(List messages) => message => + { + lock (messages) messages.Add(message); + }; + + public static bool TryCreateContext(ITestOutputHelper output, List? messages, out VulkanContext? context) + { + bool created = VulkanContext.TryCreate(ContextOptions(messages), out context, out string? failureReason); + if (!created) CheckUnavailable(failureReason, output); + else + { + try { CheckContext(context!, output); } + catch { context!.Dispose(); throw; } + } + return created; + } + + private static readonly ConditionalWeakTable> DeviceMessages = new(); + + /// + /// A device, not yet initialised, whose context will come up with the + /// suite's validation features and record every layer message for + /// . The client's own diagnostics channel still + /// receives them too. + /// + public static VulkanDevice NewDevice() + { + var messages = new List(); + // Blocking pipeline creation: these tests read pixels back after one frame, and a + // background compile would skip that frame's draw. The async path has its own tests + // (PipelineCacheTests), which set SynchronousPipelines = false before Initialize. + var device = new VulkanDevice { DebugMode = true, SynchronousPipelines = true }; + device.ConfigureContextOptions = options => + { + options.EnableValidation = true; + options.ValidationFeatures = ValidationFeatures; + options.PreferredDeviceIndex = DeviceIndex; + Action? client = options.DebugCallback; + options.DebugCallback = message => + { + lock (messages) messages.Add(message); + client?.Invoke(message); + }; + }; + DeviceMessages.Add(device, messages); + return device; + } + + public static bool TryCreateDevice(ITestOutputHelper output, out VulkanDevice? device, Action? configure = null) + { + VulkanDevice created = NewDevice(); + try + { + configure?.Invoke(created); + if (created.Initialize(IntPtr.Zero, 0, 0, out string failureReason)) + { + CheckContext(created.ContextForTests, output); + device = created; + return true; + } + CheckUnavailable(failureReason, output); + } + catch { created.Dispose(); throw; } + created.Dispose(); + device = null; + return false; + } + + /// The layer messages a device from has recorded so far. + public static List MessagesOf(VulkanDevice seam) => + seam is VulkanDevice device && DeviceMessages.TryGetValue(device, out List? messages) + ? messages + : new List(); + + /// + /// Fails on validation errors, synchronization hazards and device errors + /// such as rejected shaders or failed Vulkan calls. + /// + public static void AssertClean(VulkanDevice seam) + { + List messages = MessagesOf(seam); + ValidationAssert.NoErrors(messages); + + string? diagnostics = seam.GetError(); + if (string.IsNullOrEmpty(diagnostics)) return; + + // GetError repeats the layer messages (sanitised for the client's + // string.Format); those were judged above, so only the rest counts here. + string residual = diagnostics; + foreach (string message in ValidationAssert.Snapshot(messages)) + { + residual = residual.Replace(message.Replace('{', '[').Replace('}', ']'), ""); + } + + var remaining = new List(); + foreach (string line in residual.Split('\n')) + { + if (line.Trim().Length > 0) remaining.Add(line); + } + Assert.True(remaining.Count == 0, "device diagnostics:\n" + string.Join("\n", remaining)); + } + /// + /// A minimal shader stand-in. The client passes its own IShader and + /// IShaderProgram implementations across the seam, so the device must work + /// against the interfaces rather than any concrete type. + /// + internal sealed class TestShader : IShader + { + public EnumShaderType Type { get; set; } + public string Code { get; set; } = ""; + public string PrefixCode { get; set; } = ""; + public bool Compile() => true; + } + + internal sealed class TestProgram : IShaderProgram + { + public int ProgramId { get; set; } + public string AssetDomain { get; set; } = "game"; + public int PassId { get; set; } + public string PassName { get; set; } = "test"; + public bool ClampTexturesToEdge { get; set; } + public IShader VertexShader { get; set; } = null!; + public IShader FragmentShader { get; set; } = null!; + public IShader GeometryShader { get; set; } = null!; + public bool Oit { get; set; } = true; + public bool Disposed => false; + public bool LoadError => false; + public Vintagestory.API.Datastructures.OrderedDictionary UBOs { get; } = new(); + + public void Use() { } + public void Stop() { } + public bool Compile() => true; + public void Dispose() { } + public void Uniform(string uniformName, float value) { } + public void Uniform(string uniformName, int value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2f value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2i value) { } + public void Uniform(string uniformName, float valueX, float valueY) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec3f value) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ, float valueW) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec4f value) { } + public void Uniforms4(string uniformName, int count, float[] values) { } + public void UniformMatrix(string uniformName, float[] matrix) { } + public void BindTexture2D(string samplerName, int textureId, int textureNumber) { } + public void BindTextureCube(string samplerName, int textureId, int textureNumber) { } + public void UniformMatrices(string uniformName, int count, float[] matrix) { } + public void UniformMatrices4x3(string uniformName, int count, float[] matrix) { } + public bool HasUniform(string uniformName) => false; + } + + internal static int LinkProgram( + VulkanDevice device, string vertexCode, string fragmentCode, string name = "test") + { + var vertex = new TestShader { Type = EnumShaderType.VertexShader, Code = vertexCode }; + var fragment = new TestShader { Type = EnumShaderType.FragmentShader, Code = fragmentCode }; + + Assert.True(device.CompileShader(vertex)); + Assert.True(device.CompileShader(fragment)); + + var program = new TestProgram { PassName = name, VertexShader = vertex, FragmentShader = fragment }; + int programId = device.LinkProgram(program); + Assert.True(programId > 0, device.GetError() ?? "link failed"); + return programId; + } + +} +} + +// Source: Optimum.Render.Vulkan.Tests/SetupQueue.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; + +/// +/// Component tests' stand-in for the synchronous setup submit the renderer no +/// longer has (Phase 1B step 3 deleted VulkanCommands.SubmitAndWait). +/// +/// It owns a Transfer timeline and an for managers +/// used without a frame ring. appends the test's +/// commands to the open upload batch, after every upload recorded so far, submits +/// the batch on its own and waits for its Transfer value: the order a test wrote +/// its calls in is the order the GPU runs them. Test code only; the renderer +/// itself never waits for an upload. +/// +internal sealed unsafe class SetupQueue : IDisposable +{ + private readonly VulkanContext _context; + private bool _disposed; + + public FrameTimeline Timeline { get; } + public RetireQueue Retired { get; } + public UploadManager Uploads { get; } + + public SetupQueue(VulkanContext context, ulong stagingPerSlot = 4UL << 20) + { + _context = context; + Timeline = new FrameTimeline(context); + Retired = new RetireQueue(Timeline); + Uploads = new UploadManager(context, Timeline, Retired, framesInFlight: 2, stagingPerSlot); + } + + /// Records into the open upload batch, submits it and waits for it. + public void SubmitAndWait(Action record) + { + CommandBuffer commandBuffer = Uploads.BeginRecording(inlineInFrame: false); + try + { + record(commandBuffer); + } + finally + { + Uploads.EndRecording(); + } + + ulong transferValue = Uploads.SubmitStandalone(); + Timeline.WaitForTransfer(transferValue, WaitSite.Readback); + Retired.Collect(); + } + + /// + /// Moves a standalone image between layouts with a broad synchronization2 + /// barrier (all commands on both sides), for tests that drive raw images. + /// + public void TransitionImage(CommandBuffer commandBuffer, VulkanImage image, ImageLayout target, ImageAspectFlags aspect) + { + var barrier = new ImageMemoryBarrier2 + { + SType = StructureType.ImageMemoryBarrier2, + SrcStageMask = PipelineStageFlags2.AllCommandsBit, + SrcAccessMask = AccessFlags2.MemoryWriteBit, + DstStageMask = PipelineStageFlags2.AllCommandsBit, + DstAccessMask = AccessFlags2.MemoryReadBit | AccessFlags2.MemoryWriteBit, + OldLayout = image.Layout, + NewLayout = target, + Image = image.Handle, + SubresourceRange = new ImageSubresourceRange(aspect, 0, 1, 0, 1), + }; + + var dependency = new DependencyInfo + { + SType = StructureType.DependencyInfo, + ImageMemoryBarrierCount = 1, + PImageMemoryBarriers = &barrier, + }; + + _context.Api.CmdPipelineBarrier2(commandBuffer, &dependency); + image.Layout = target; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + Uploads.Dispose(); + Retired.DisposeAll(); + Timeline.Dispose(); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/Graph/FrameExecutionTests.cs b/Optimum.Render.Vulkan.Tests/Graph/FrameExecutionTests.cs new file mode 100644 index 00000000..7477f62c --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Graph/FrameExecutionTests.cs @@ -0,0 +1,891 @@ +// Source: Optimum.Render.Vulkan.Tests/ComputePassTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +/// +/// The frame graph's compute pass kind on a device, validation with sync and best +/// practices on: a dispatch stores into a storage image in the format the device +/// supports (or its fallback) with its specialization constants and push constants; a +/// mip chain samples level n and stores level n + 1 of one image; a compute pass +/// followed by a raster pass sampling its output, frame after frame with Present +/// between frames and no readback in the loop; and a declaration that does not fit is +/// refused without recording anything. +/// +public class ComputePassTests(ITestOutputHelper output) +{ + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + private static string Qualifier(Format format) => format switch + { + Format.R8Unorm => "r8", + Format.R8G8Unorm => "rg8", + Format.R8G8B8A8Unorm => "rgba8", + _ => throw new ArgumentOutOfRangeException(nameof(format), format, null), + }; + + private static int Channels(Format format) => format switch + { + Format.R8Unorm => 1, + Format.R8G8Unorm => 2, + _ => 4, + }; + + private int Program(VulkanDevice seam, string code, string name, ComputeSlot[] slots, uint push = 0) + { + int program = seam.CreateComputeProgram(code, name, slots, push); + Assert.True(program > 0, seam.GetError()); + return program; + } + + [SkippableFact] + public void ADispatchStoresIntoAStorageImageWithItsSpecializationAndPushConstants() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + const int width = 13, height = 11; + + // The AO working term's format: the device's choice, or its fallback. + int target = seam.CreateStorageTexture(width, height, Format.R8Unorm); + VulkanTexture texture = seam.TexturesForTests.Get(target)!; + Assert.Equal(StorageFormats.Choose(Format.R8Unorm, seam.ContextForTests.OptimalFormatFeatures), texture.Format); + Assert.NotEqual((ImageUsageFlags)0, texture.Usage & ImageUsageFlags.StorageBit); + output.WriteLine("R8_UNORM storage texture created as " + texture.Format); + int channels = Channels(texture.Format); + + string code = $$""" + #version 450 + layout(local_size_x = 8, local_size_y = 8) in; + layout(constant_id = 0) const uint LEVEL = 0u; + layout(push_constant) uniform Push { uint offset; } push; + layout(binding = 3, {{Qualifier(texture.Format)}}) uniform writeonly image2D target; + void main() + { + ivec2 p = ivec2(gl_GlobalInvocationID.xy); + if (any(greaterThanEqual(p, imageSize(target)))) return; + uint value = (uint(p.x) * 16u + uint(p.y) + LEVEL + push.offset) & 255u; + imageStore(target, p, vec4(float(value) / 255.0)); + } + """; + int program = Program(seam, code, "store", new[] { new ComputeSlot(3, ComputeSlotKind.Storage) }, 4); + + uint[] levels = { 7, 100, 7 }; + long dispatchesBefore = seam.FrameGraphForTests.Dispatches; + for (int frame = 0; frame < levels.Length; frame++) + { + seam.BeginFrame(); + uint offset = (uint)frame * 3; + Assert.True(seam.RecordComputePass(new ComputePassDeclaration + { + Name = "store", + ProgramId = program, + Specialization = new[] { levels[frame] }, + Bindings = new[] { new ComputeBinding(3, target, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0, BitConverter.GetBytes(offset)) }, + }), seam.GetError()); + + byte[] pixels = seam.ReadBackLevel0ForTests(target); + for (int y = 0; y < height; y++) + for (int x = 0; x < width; x++) + { + int expected = (int)((x * 16 + y + levels[frame] + offset) & 255); + Assert.Equal(expected, pixels[(y * width + x) * channels]); + } + seam.Present(); + } + + // Two distinct specialization values: two pipelines, the third frame a hit. + Assert.Equal(2, seam.ComputeForTests.Get(program)!.PipelineCount); + Assert.Equal(1, seam.ComputeForTests.Hits); + // 13x11 at 8x8 is one dispatch of 2x2 groups per frame. + Assert.Equal(levels.Length, seam.FrameGraphForTests.Dispatches - dispatchesBefore); + GpuTest.AssertClean(seam); + } + } + + [SkippableFact] + public void AMipChainSamplesLevelNAndStoresLevelNPlusOneOfOneImage() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + const int size = 16, levels = 3; + int chain = seam.CreateStorageTexture(size, size, Format.R8G8B8A8Unorm, levels); + Assert.Equal((uint)levels, seam.TexturesForTests.Get(chain)!.MipLevels); + + int seed = Program(seam, """ + #version 450 + layout(local_size_x = 8, local_size_y = 8) in; + layout(push_constant) uniform Push { uint salt; } push; + layout(binding = 0, rgba8) uniform writeonly image2D level0; + void main() + { + ivec2 p = ivec2(gl_GlobalInvocationID.xy); + if (any(greaterThanEqual(p, imageSize(level0)))) return; + uint value = (uint(p.x) * 37u + uint(p.y) * 11u + push.salt) & 255u; + imageStore(level0, p, vec4(float(value) / 255.0, 0.0, 0.0, 1.0)); + } + """, "seed", new[] { new ComputeSlot(0, ComputeSlotKind.Storage) }, 4); + + // The prefilter shape: the sampled binding's view starts at level n, so lod 0 is level n. + int reduce = Program(seam, """ + #version 450 + layout(local_size_x = 8, local_size_y = 8) in; + layout(binding = 0) uniform sampler2D source; + layout(binding = 1, rgba8) uniform writeonly image2D destination; + void main() + { + ivec2 p = ivec2(gl_GlobalInvocationID.xy); + if (any(greaterThanEqual(p, imageSize(destination)))) return; + float m = 0.0; + for (int i = 0; i < 4; i++) m = max(m, texelFetch(source, 2 * p + ivec2(i & 1, i >> 1), 0).r); + imageStore(destination, p, vec4(m, 0.0, 0.0, 1.0)); + } + """, "reduce", new[] + { + new ComputeSlot(0, ComputeSlotKind.Sampled), new ComputeSlot(1, ComputeSlotKind.Storage), + }); + + for (uint frame = 0; frame < 3; frame++) + { + seam.BeginFrame(); + uint salt = frame * 29; + Assert.True(seam.RecordComputePass(new ComputePassDeclaration + { + Name = "seed", + ProgramId = seed, + Bindings = new[] { new ComputeBinding(0, chain, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0, BitConverter.GetBytes(salt)) }, + }), seam.GetError()); + for (uint n = 0; n + 1 < levels; n++) + { + Assert.True(seam.RecordComputePass(new ComputePassDeclaration + { + Name = "reduce", + ProgramId = reduce, + Bindings = new[] + { + new ComputeBinding(0, chain, ComputeAccess.Sampled, BaseMip: n), + new ComputeBinding(1, chain, ComputeAccess.StorageWrite, BaseMip: n + 1), + }, + Dispatches = new[] { ComputeDispatch.Covering(1) }, + }), seam.GetError()); + } + + // Only after the whole chain is recorded: the expected chain on the CPU. + var expected = new int[levels][]; + expected[0] = new int[size * size]; + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + expected[0][y * size + x] = (int)((x * 37 + y * 11 + salt) & 255); + for (int n = 1; n < levels; n++) + { + int s = size >> n, parent = size >> (n - 1); + expected[n] = new int[s * s]; + for (int y = 0; y < s; y++) + for (int x = 0; x < s; x++) + { + int m = 0; + for (int i = 0; i < 4; i++) + m = Math.Max(m, expected[n - 1][(2 * y + (i >> 1)) * parent + 2 * x + (i & 1)]); + expected[n][y * s + x] = m; + } + } + + for (uint n = 0; n < levels; n++) + { + int s = size >> (int)n; + byte[] pixels = seam.ReadBackLevelForTests(chain, n); + Assert.Equal(s * s * 4, pixels.Length); + for (int i = 0; i < s * s; i++) + { + Assert.Equal(expected[n][i], pixels[i * 4]); + Assert.Equal(255, pixels[i * 4 + 3]); + } + } + seam.Present(); + } + GpuTest.AssertClean(seam); + } + } + + [SkippableFact] + public void ARasterPassSamplesTheComputePassOutputOfTheSameFrameFrameAfterFrame() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + const int size = 8, frames = 8; + + int computed = seam.CreateStorageTexture(size, size, Format.R8G8B8A8Unorm); + int store = Program(seam, """ + #version 450 + layout(local_size_x = 4, local_size_y = 4) in; + layout(push_constant) uniform Push { uint frame; } push; + layout(binding = 0, rgba8) uniform writeonly image2D target; + void main() + { + ivec2 p = ivec2(gl_GlobalInvocationID.xy); + if (any(greaterThanEqual(p, imageSize(target)))) return; + imageStore(target, p, vec4(float((push.frame + 1u) * 20u) / 255.0, float(p.x * 16) / 255.0, + float(p.y * 16) / 255.0, 1.0)); + } + """, "store", new[] { new ComputeSlot(0, ComputeSlotKind.Storage) }, 4); + + int background = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + layout(location = 0) out vec4 outColor; + void main(void) { outColor = vec4(1.0, 0.0, 1.0, 1.0); } + """, "background"); + int sample = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D computed; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = texelFetch(computed, ivec2(gl_FragCoord.xy), 0); } + """, "sample"); + + var colours = new int[frames]; + var targets = new int[frames]; + for (int i = 0; i < frames; i++) + { + colours[i] = seam.CreateTexture2DRaw(size, size, 0x8058, IntPtr.Zero, 4); + targets[i] = seam.CreateFramebuffer(size, size); + seam.AttachTexture(targets[i], EnumFramebufferAttachment.ColorAttachment0, colours[i], 0); + seam.SetDrawBuffers(targets[i], 1); + } + + seam.SetSamplerUnit(sample, "computed", 0); + long passesBefore = seam.FrameGraphForTests.ComputePasses; + long dispatchesBefore = seam.FrameGraphForTests.Dispatches; + + for (int i = 0; i < frames; i++) + { + seam.BeginFrame(); + // A draw first, so the frame has a rendering scope open when the compute pass comes. + seam.BindFramebuffer(targets[i]); + seam.SetViewport(0, 0, size, size); + seam.SetCullFace(false); + seam.SetDepthTest(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.UseProgram(background); + seam.DrawFullscreenTriangle(); + + Assert.True(seam.RecordComputePass(new ComputePassDeclaration + { + Name = "store", + ProgramId = store, + Bindings = new[] { new ComputeBinding(0, computed, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0, BitConverter.GetBytes((uint)i)) }, + }), seam.GetError()); + + seam.UseProgram(sample); + seam.BindTexture(0, computed); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + seam.Present(); + } + + // The work group size comes from the module (4x4), not the description's default 8x8. + ComputeProgram stored = seam.ComputeForTests.Get(store)!; + Assert.Equal((4u, 4u), (stored.LocalSizeX, stored.LocalSizeY)); + Assert.Equal(frames, seam.FrameGraphForTests.ComputePasses - passesBefore); + Assert.Equal(frames, seam.FrameGraphForTests.Dispatches - dispatchesBefore); + + // The sequence completes before any readback or CPU wait. + seam.BeginFrame(); + for (int i = 0; i < frames; i++) + { + byte[] pixels = seam.ReadBackLevel0ForTests(colours[i]); + for (int y = 0; y < size; y++) + for (int x = 0; x < size; x++) + { + int offset = (y * size + x) * 4; + Assert.Equal((i + 1) * 20, pixels[offset]); + Assert.Equal(x * 16, pixels[offset + 1]); + Assert.Equal(y * 16, pixels[offset + 2]); + Assert.Equal(255, pixels[offset + 3]); + } + } + seam.Present(); + GpuTest.AssertClean(seam); + } + } + + [SkippableFact] + public void ADeclarationThatDoesNotFitItsProgramIsRefusedWithoutRecording() + { + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No Vulkan device"); + using (device) + { + VulkanDevice seam = device!; + int plain = seam.CreateTexture2DRaw(4, 4, 0x8058, IntPtr.Zero, 4); + int storage = seam.CreateStorageTexture(4, 4, Format.R8G8B8A8Unorm, 2); + int program = Program(seam, """ + #version 450 + layout(local_size_x = 4, local_size_y = 4) in; + layout(binding = 0, rgba8) uniform writeonly image2D target; + void main() { imageStore(target, ivec2(gl_GlobalInvocationID.xy), vec4(1.0)); } + """, "store", new[] { new ComputeSlot(0, ComputeSlotKind.Storage) }); + + Assert.False(seam.RecordComputePass(new ComputePassDeclaration + { + ProgramId = program, + Bindings = new[] { new ComputeBinding(0, storage, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0) }, + }), "no frame is open"); + + long passesBefore = seam.FrameGraphForTests.ComputePasses; + seam.BeginFrame(); + void Refused(ComputePassDeclaration pass, string reason) + { + Assert.False(seam.RecordComputePass(pass)); + Assert.Contains(reason, seam.GetError()); + } + Refused(new ComputePassDeclaration + { + Name = "plain", ProgramId = program, + Bindings = new[] { new ComputeBinding(0, plain, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0) }, + }, "not created as a storage texture"); + Refused(new ComputePassDeclaration + { + Name = "kind", ProgramId = program, + Bindings = new[] { new ComputeBinding(0, storage, ComputeAccess.Sampled) }, + Dispatches = new[] { ComputeDispatch.Covering(0) }, + }, "is declared Storage"); + Refused(new ComputePassDeclaration + { + Name = "unbound", ProgramId = program, + Bindings = Array.Empty(), + Dispatches = new[] { ComputeDispatch.Explicit(1, 1) }, + }, "binding 0 is not bound"); + Refused(new ComputePassDeclaration + { + Name = "push", ProgramId = program, + Bindings = new[] { new ComputeBinding(0, storage, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0, new byte[4]) }, + }, "pushes 4 bytes"); + Refused(new ComputePassDeclaration + { + Name = "missing", ProgramId = 999, + Bindings = new[] { new ComputeBinding(0, storage, ComputeAccess.StorageWrite) }, + Dispatches = new[] { ComputeDispatch.Covering(0) }, + }, "no compute program 999"); + Assert.Equal(passesBefore, seam.FrameGraphForTests.ComputePasses); + Assert.Equal(0, seam.CreateComputeProgram("#version 450\nvoid main() { broken }", "broken", + Array.Empty())); + Assert.Contains("failed to compile", seam.GetError()); + seam.Present(); + + // A deleted program retires on the timeline, after the frames that could bind it. + seam.DeleteComputeProgram(program); + Assert.Null(seam.ComputeForTests.Get(program)); + GpuTest.AssertClean(seam); + } + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/FrameGraphFrameTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; +using static Optimum.Render.Vulkan.Tests.GpuTest; + +/// +/// Phase 2 step 2 on a real device: the streaming frame graph. A declared TAA-shaped +/// frame (opaque scene with a motion window, history resolve, final composition that +/// writes Primary 0 while sampling Primary 1, blit) records exactly one rendering scope +/// per pass for five frames with the history accumulating (Present between frames, no +/// readback in the loop); the same frames are byte-identical with +/// OPTIMUM_VULKAN_FRAMEGRAPH off; clears are promoted, kept in the pass or landed +/// standalone as the plan describes, and a masked-out clear stays a no-op. +/// +public class FrameGraphFrameTests +{ + private readonly ITestOutputHelper _output; + + public FrameGraphFrameTests(ITestOutputHelper output) => _output = output; + + private const int Size = 8; + + private const string FullscreenVertex = """ + #version 330 core + out vec2 uv; + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.5, 1.0); + uv = vec2((x + 1.0) * 0.5, (y + 1.0) * 0.5); + } + """; + + private sealed class TaaScene + { + public int Primary, Colour, Glow, Motion, Depth; + public int HistoryAFb, HistoryA, HistoryBFb, HistoryB; + public int OutputFb, Output; + public int Scene, Resolve, Final, Blit; + } + + private bool TryCreate(bool frameGraph, out VulkanDevice? device) + { + VulkanDevice created = NewDevice(); + created.FrameGraphEnabled = frameGraph; + if (!created.Initialize(IntPtr.Zero, 0, 0, out string failureReason)) + { + _output.WriteLine("Vulkan unavailable: " + failureReason); + created.Dispose(); + device = null; + return false; + } + device = created; + return true; + } + + private static TaaScene CreateTaaScene(VulkanDevice seam) + { + int Texture(EnumTextureInternalFormat format) => + seam.CreateTexture2D(Size, Size, format, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + + var s = new TaaScene + { + Colour = Texture(EnumTextureInternalFormat.Rgba8), + Glow = Texture(EnumTextureInternalFormat.Rgba8), + Motion = Texture(EnumTextureInternalFormat.Rgba16f), + Depth = Texture(EnumTextureInternalFormat.DepthComponent32), + HistoryA = Texture(EnumTextureInternalFormat.Rgba16f), + HistoryB = Texture(EnumTextureInternalFormat.Rgba16f), + Output = Texture(EnumTextureInternalFormat.Rgba8), + }; + s.Primary = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(s.Primary, EnumFramebufferAttachment.ColorAttachment0, s.Colour, 0); + seam.AttachTexture(s.Primary, EnumFramebufferAttachment.ColorAttachment1, s.Glow, 0); + seam.AttachTexture(s.Primary, EnumFramebufferAttachment.ColorAttachment2, s.Motion, 0); + seam.AttachTexture(s.Primary, EnumFramebufferAttachment.DepthAttachment, s.Depth, 0); + seam.SetDrawBuffers(s.Primary, 0b011); + s.HistoryAFb = SingleTarget(seam, s.HistoryA); + s.HistoryBFb = SingleTarget(seam, s.HistoryB); + s.OutputFb = SingleTarget(seam, s.Output); + + s.Scene = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + in vec2 uv; + layout(location = 0) out vec4 outColor; + layout(location = 1) out vec4 outGlow; + layout(location = 2) out vec4 outMotion; + void main(void) + { + outColor = vec4(1.0, 0.2, 0.0, 1.0); + outGlow = vec4(0.4, 0.0, 0.0, 1.0); + outMotion = vec4(0.25, 0.5, 0.0, 1.0); + } + """, "fg-scene"); + s.Resolve = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D sceneTex; + uniform sampler2D historyTex; + uniform sampler2D motionTex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) + { + outColor = vec4(mix(texture(historyTex, uv).r, texture(sceneTex, uv).r, 0.5), + texture(motionTex, uv).g, 0.0, 1.0); + } + """, "fg-resolve"); + s.Final = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D glowTex; + uniform sampler2D historyTex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = vec4(texture(historyTex, uv).r, texture(glowTex, uv).r, 0.2, 1.0); } + """, "fg-final"); + s.Blit = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D sceneTex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = texture(sceneTex, uv); } + """, "fg-blit"); + seam.SetSamplerUnit(s.Resolve, "sceneTex", 0); + seam.SetSamplerUnit(s.Resolve, "historyTex", 1); + seam.SetSamplerUnit(s.Resolve, "motionTex", 2); + seam.SetSamplerUnit(s.Final, "glowTex", 0); + seam.SetSamplerUnit(s.Final, "historyTex", 1); + seam.SetSamplerUnit(s.Blit, "sceneTex", 0); + return s; + } + + private static int SingleTarget(VulkanDevice seam, int texture) + { + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(framebuffer, 1); + return framebuffer; + } + + private static void BaseState(VulkanDevice seam) + { + seam.SetViewport(0, 0, Size, Size); + seam.SetScissorEnabled(false); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.SetColorMask(true, true, true, true); + } + + /// Frame 0: both histories cleared, nothing drawn. + private static void SeedFrame(VulkanDevice seam, TaaScene s) + { + seam.BeginFrame(); + BaseState(seam); + seam.DeclarePass(new PassDeclaration { Name = "Seed", FramebufferId = s.HistoryAFb }); + seam.ClearColor(0, 0f, 0f, 0f, 0f); + seam.DeclarePass(new PassDeclaration { Name = "Seed", FramebufferId = s.HistoryBFb }); + seam.ClearColor(0, 0f, 0f, 0f, 0f); + seam.Present(); + } + + /// + /// One TAA-shaped frame. Odd frames write history A and read B, even frames the reverse. + /// Without TAA the resolve is skipped and the final pass reads the seeded history A. + /// + private static void TaaFrame(VulkanDevice seam, TaaScene s, int frame, bool taa) + { + bool writeA = frame % 2 == 1; + int writeFb = writeA ? s.HistoryAFb : s.HistoryBFb; + int writeTex = taa ? writeA ? s.HistoryA : s.HistoryB : s.HistoryA; + int readTex = writeA ? s.HistoryB : s.HistoryA; + + seam.BeginFrame(); + BaseState(seam); + + // Opaque: every clear issued before the first draw, the motion one inside a motion window. + seam.DeclarePass(new PassDeclaration { Name = "Opaque", FramebufferId = s.Primary }); + seam.SetViewport(0, 0, Size, Size); + seam.SetDrawBuffers(s.Primary, 0b011); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearColor(1, 0f, 0f, 0f, 1f); + seam.SetDrawBuffers(s.Primary, 0b111); + seam.ClearColor(2, 0f, 0f, 0f, 0f); + seam.SetDrawBuffers(s.Primary, 0b011); + seam.SetDepthMask(true); + seam.ClearDepth(1f); + seam.SetDepthTest(true); + seam.SetDepthFunc(0x203); + seam.UseProgram(s.Scene); + seam.SetDrawBuffers(s.Primary, 0b111); + seam.DrawFullscreenTriangle(); + seam.SetDrawBuffers(s.Primary, 0b011); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + + if (taa) + { + seam.DeclarePass(new PassDeclaration + { + Name = "TaaResolve", FramebufferId = writeFb, Reads = new[] { s.Colour, readTex, s.Motion }, + }); + seam.SetViewport(0, 0, Size, Size); + seam.UseProgram(s.Resolve); + seam.BindTexture(0, s.Colour); + seam.BindTexture(1, readTex); + seam.BindTexture(2, s.Motion); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + seam.BindTexture(1, 0); + seam.BindTexture(2, 0); + } + + // Final composition: writes Primary 0, samples Primary 1. + seam.DeclarePass(new PassDeclaration + { + Name = "FinalComposition", FramebufferId = s.Primary, ColorSlots = ~(1u << 1), + Reads = new[] { s.Glow, writeTex }, + }); + seam.SetViewport(0, 0, Size, Size); + seam.SetDrawBuffers(s.Primary, 0b001); + seam.UseProgram(s.Final); + seam.BindTexture(0, s.Glow); + seam.BindTexture(1, writeTex); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + seam.BindTexture(1, 0); + seam.EndPass(); + seam.SetDrawBuffers(s.Primary, 0b011); + + // Blit: a plain full overwrite, so an exactly matching plan loads it DONT_CARE. + seam.DeclarePass(new PassDeclaration + { + Name = "Blit", FramebufferId = s.OutputFb, Reads = new[] { s.Colour }, TransientSlots = 1, + }); + seam.SetViewport(0, 0, Size, Size); + seam.UseProgram(s.Blit); + seam.BindTexture(0, s.Colour); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + + seam.Present(); + } + + private static Dictionary ReadAll(VulkanDevice seam, TaaScene s) + { + seam.BeginFrame(); + var result = new Dictionary + { + ["colour"] = seam.ReadBackLevel0ForTests(s.Colour), + ["glow"] = seam.ReadBackLevel0ForTests(s.Glow), + ["motion"] = seam.ReadBackLevel0ForTests(s.Motion), + ["depth"] = seam.ReadBackLevel0ForTests(s.Depth), + ["historyA"] = seam.ReadBackLevel0ForTests(s.HistoryA), + ["historyB"] = seam.ReadBackLevel0ForTests(s.HistoryB), + ["output"] = seam.ReadBackLevel0ForTests(s.Output), + }; + seam.Present(); + return result; + } + + private static float HalfAt(byte[] texels, int pixel, int channel) => + (float)BitConverter.ToHalf(texels, pixel * 8 + channel * 2); + + [SkippableFact] + public void ADeclaredTaaFrameOpensOneScopePerPassAndAccumulatesHistory() + { + Skip.IfNot(TryCreate(frameGraph: true, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + FrameGraph graph = seam.FrameGraphForTests; + TaaScene scene = CreateTaaScene(seam); + + long standaloneBefore = graph.StandaloneClears; + SeedFrame(seam, scene); + // Nothing attached the seeded histories this frame: the clears landed as clear-image commands. + Assert.Equal(2, graph.StandaloneClears - standaloneBefore); + + long hitsBefore = graph.PlanHits; + long missesBefore = graph.PlanMisses; + long dontCareBefore = graph.PlannedDontCareLoads; + for (int frame = 1; frame <= 5; frame++) + { + long scopes = seam.ScopesOpenedForTests; + long passes = graph.Passes; + long splits = graph.Splits; + long promoted = graph.PromotedClears; + long inPass = graph.InPassClears; + long standalone = graph.StandaloneClears; + + TaaFrame(seam, scene, frame, taa: true); + + long scopesOpened = seam.ScopesOpenedForTests - scopes; + long passCount = graph.Passes - passes; + _output.WriteLine($"frame {frame}: passes={passCount} scopes={scopesOpened} splits={graph.Splits - splits} " + + $"promoted={graph.PromotedClears - promoted} inPass={graph.InPassClears - inPass} " + + $"standalone={graph.StandaloneClears - standalone}"); + Assert.Equal(4, passCount); + Assert.Equal(passCount, scopesOpened); + Assert.Equal(0, graph.Splits - splits); + // Colour, glow, motion and depth: every Opaque clear became LOAD_OP_CLEAR. + Assert.Equal(4, graph.PromotedClears - promoted); + Assert.Equal(0, graph.InPassClears - inPass); + Assert.Equal(0, graph.StandaloneClears - standalone); + } + + // Frames 1 and 2 have no frame of the same parity to match; 3 to 5 do. + Assert.Equal(3, graph.PlanHits - hitsBefore); + Assert.Equal(2, graph.PlanMisses - missesBefore); + Assert.Equal(3, graph.PlannedDontCareLoads - dontCareBefore); + + Dictionary texels = ReadAll(seam, scene); + int centre = Size / 2 * Size + Size / 2; + // h_n = (h_(n-1) + 1) / 2 from 0: A was written on frames 1, 3, 5, B on 2 and 4. + Assert.Equal(0.96875f, HalfAt(texels["historyA"], centre, 0)); + Assert.Equal(0.9375f, HalfAt(texels["historyB"], centre, 0)); + Assert.Equal(0.5f, HalfAt(texels["historyA"], centre, 1)); + // Final composition read the history it just wrote and Primary 1 while writing Primary 0. + _output.WriteLine($"centre: colour={texels["colour"][centre * 4]},{texels["colour"][centre * 4 + 1]},{texels["colour"][centre * 4 + 2]} " + + $"glow={texels["glow"][centre * 4]},{texels["glow"][centre * 4 + 1]},{texels["glow"][centre * 4 + 2]} " + + $"output={texels["output"][centre * 4]},{texels["output"][centre * 4 + 1]},{texels["output"][centre * 4 + 2]}"); + Assert.Equal(247, texels["output"][centre * 4]); + Assert.Equal(102, texels["output"][centre * 4 + 1]); + Assert.Equal(51, texels["output"][centre * 4 + 2]); + Assert.Equal(texels["colour"], texels["output"]); + AssertClean(seam); + } + } + + [SkippableTheory] + [InlineData(true)] + [InlineData(false)] + public void TheDeclaredFrameIsPixelIdenticalWithTheFrameGraphOff(bool taa) + { + Dictionary? on = RenderFrames(frameGraph: true, taa); + Skip.If(on == null, "No usable Vulkan device."); + Dictionary off = RenderFrames(frameGraph: false, taa)!; + + foreach ((string name, byte[] texels) in on!) + { + Assert.True(texels.AsSpan().SequenceEqual(off[name]), name + " differs between the frame graph and scope inference"); + } + } + + private Dictionary? RenderFrames(bool frameGraph, bool taa) + { + if (!TryCreate(frameGraph, out VulkanDevice? device)) return null; + using (device) + { + VulkanDevice seam = device!; + TaaScene scene = CreateTaaScene(seam); + SeedFrame(seam, scene); + for (int frame = 1; frame <= 5; frame++) TaaFrame(seam, scene, frame, taa); + Dictionary texels = ReadAll(seam, scene); + _output.WriteLine((frameGraph ? "graph" : "inference") + ": scopes=" + seam.ScopesOpenedForTests + + " passes=" + seam.FrameGraphForTests.Passes); + AssertClean(seam); + return texels; + } + } + + /// + /// One frame of clears on two targets: a masked-out clear and an all-false colour-mask + /// clear (no-ops, nothing pending), a clear with no pass open under an additive draw + /// (LOAD_OP_CLEAR), a clear after the pass opened (vkCmdClearAttachments, counted) and a + /// clear of a texture sampled before any pass attaches it (a clear-image command first). + /// + [SkippableTheory] + [InlineData(true)] + [InlineData(false)] + public void ClearsArePromotedKeptInThePassOrLandedBeforeARead(bool frameGraph) + { + Skip.IfNot(TryCreate(frameGraph, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + FrameGraph graph = seam.FrameGraphForTests; + int Texture() => seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int c0 = Texture(), c1 = Texture(), source = Texture(), sink = Texture(); + int target = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, c0, 0); + seam.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment1, c1, 0); + int sourceFb = SingleTarget(seam, source); + int sinkFb = SingleTarget(seam, sink); + + int add = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + layout(location = 0) out vec4 outColor; + void main(void) { outColor = vec4(0.4, 0.0, 0.0, 0.0); } + """, "fg-add"); + int copy = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D tex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = texture(tex, uv); } + """, "fg-copy"); + seam.SetSamplerUnit(copy, "tex", 0); + + // Seed: c1 grey. The target has both draw buffers. + seam.BeginFrame(); + BaseState(seam); + seam.DeclarePass(new PassDeclaration { Name = "Seed", FramebufferId = target }); + seam.SetDrawBuffers(target, 0b11); + seam.ClearColor(1, 0.2f, 0.2f, 0.2f, 1f); + seam.Present(); + + seam.BeginFrame(); + BaseState(seam); + seam.DeclarePass(new PassDeclaration { Name = "Target", FramebufferId = target }); + seam.SetDrawBuffers(target, 0b01); + + (long promoted, long inPass, long standalone) Counters() => + (graph.PromotedClears, graph.InPassClears, graph.StandaloneClears); + var before = Counters(); + // Masked out by the draw buffers, then by an all-false colour mask: no-ops on every path. + seam.ClearColor(1, 1f, 1f, 1f, 1f); + seam.SetColorMask(false, false, false, false); + seam.ClearColor(0, 1f, 1f, 1f, 1f); + seam.SetColorMask(true, true, true, true); + Assert.Equal(before, Counters()); + Assert.False(graph.HasPendingClears); + + // No pass open: promoted into the scope the additive draw opens. + long scopes = seam.ScopesOpenedForTests; + seam.ClearColor(0, 0.2f, 0f, 0f, 1f); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetBlendFuncSeparate(0, 1, 1, 1, 1); + seam.UseProgram(add); + seam.DrawFullscreenTriangle(); + seam.SetBlend(false, EnumBlendMode.Standard); + Assert.Equal(1, seam.ScopesOpenedForTests - scopes); + + // The pass is open: this clear stays in it. + seam.SetDrawBuffers(target, 0b11); + seam.ClearColor(1, 0f, 1f, 0f, 1f); + seam.SetDrawBuffers(target, 0b01); + if (frameGraph) + { + Assert.Equal(before.promoted + 1, graph.PromotedClears); + Assert.Equal(before.inPass + 1, graph.InPassClears); + Assert.Equal(before.standalone, graph.StandaloneClears); + } + + // Cleared with no pass open, then sampled by a pass that does not attach it. + seam.DeclarePass(new PassDeclaration { Name = "ClearSource", FramebufferId = sourceFb }); + seam.ClearColor(0, 0f, 0f, 1f, 1f); + seam.DeclarePass(new PassDeclaration { Name = "Sink", FramebufferId = sinkFb, Reads = new[] { source } }); + seam.UseProgram(copy); + seam.BindTexture(0, source); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + if (frameGraph) + { + Assert.Equal(before.standalone + 1, graph.StandaloneClears); + Assert.Equal(0, graph.Splits); + } + seam.Present(); + + seam.BeginFrame(); + byte[] first = seam.ReadBackLevel0ForTests(c0); + byte[] second = seam.ReadBackLevel0ForTests(c1); + byte[] sampled = seam.ReadBackLevel0ForTests(sink); + seam.Present(); + + int centre = (Size / 2 * Size + Size / 2) * 4; + // 0.2 cleared + 0.4 added = 0.6 (153); alpha 1 + 0. + Assert.Equal(new byte[] { 153, 0, 0, 255 }, first.AsSpan(centre, 4).ToArray()); + Assert.Equal(new byte[] { 0, 255, 0, 255 }, second.AsSpan(centre, 4).ToArray()); + Assert.Equal(new byte[] { 0, 0, 255, 255 }, sampled.AsSpan(centre, 4).ToArray()); + AssertClean(seam); + } + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/MotionWindowTests.cs b/Optimum.Render.Vulkan.Tests/MotionWindowTests.cs new file mode 100644 index 00000000..08f22f76 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/MotionWindowTests.cs @@ -0,0 +1,384 @@ +using System; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Xunit; +using Xunit.Abstractions; +using static Optimum.Render.Vulkan.Tests.MotionFixture; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// Phase 2 contract C4 on a real device, once per colour write tier (forced the +/// way OPTIMUM_VULKAN_COLOR_WRITE_TIER forces it, through the context options): +/// draw buffers and the TAA motion windows are write masks inside one rendering +/// scope, never scope restarts. Every test reads back only after the draws, in +/// the frame, and asserts bytes rather than approximations. +/// +public class MotionWindowTests +{ + private readonly ITestOutputHelper _output; + + public MotionWindowTests(ITestOutputHelper output) => _output = output; + + private const int Size = 8; + + private const string FullscreenVertex = """ + #version 330 core + out vec2 uv; + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + uv = vec2((x + 1.0) * 0.5, (y + 1.0) * 0.5); + } + """; + + private static string FourOutputs(string colour, string glow, string motion, string extra) => $$""" + #version 330 core + layout(location = 0) out vec4 outColor; + layout(location = 1) out vec4 outGlow; + layout(location = 2) out vec4 outMotion; + layout(location = 3) out vec4 outExtra; + void main(void) + { + outColor = vec4({{colour}}); + outGlow = vec4({{glow}}); + outMotion = vec4({{motion}}); + outExtra = vec4({{extra}}); + } + """; + + /// The OPTIMUM_VULKAN_COLOR_WRITE_TIER tokens. + public static TheoryData Tiers => new() { "enable", "mask", "pipeline" }; + + /// Every tier with the frame graph on and off (OPTIMUM_VULKAN_FRAMEGRAPH). + public static TheoryData TiersWithFrameGraph => new() + { + { "enable", true }, { "mask", true }, { "pipeline", true }, + { "enable", false }, { "mask", false }, { "pipeline", false }, + }; + + private static ColorWriteTier Tier(string token) => + DeviceCaps.ParseColorWriteTier(token) ?? throw new ArgumentException("unknown tier " + token); + + private bool TryCreateDevice(ColorWriteTier tier, out VulkanDevice? device) + { + VulkanDevice created = GpuTest.NewDevice(); + Action? suite = created.ConfigureContextOptions; + created.ConfigureContextOptions = options => + { + suite?.Invoke(options); + options.ColorWriteTier = tier; + }; + + if (!created.Initialize(IntPtr.Zero, 0, 0, out string failureReason)) + { + _output.WriteLine("Vulkan unavailable: " + failureReason); + created.Dispose(); + device = null; + return false; + } + if (created.ColorWriteTierForTests != tier) + { + _output.WriteLine("device lacks tier " + tier + "; selected " + created.ColorWriteTierForTests); + created.Dispose(); + device = null; + return false; + } + device = created; + return true; + } + + private sealed class Scene + { + public int Framebuffer; + public int Colour; + public int Glow; + public int Motion; + public int Extra; + } + + /// Colour and glow RGBA8, motion RGBA16F, a fourth RGBA8 slot: Primary's shape with TAA on. + private static Scene CreateScene(VulkanDevice seam) + { + var scene = new Scene + { + Colour = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + Glow = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + Motion = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + Extra = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }; + scene.Framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(scene.Framebuffer, EnumFramebufferAttachment.ColorAttachment0, scene.Colour, 0); + seam.AttachTexture(scene.Framebuffer, EnumFramebufferAttachment.ColorAttachment1, scene.Glow, 0); + seam.AttachTexture(scene.Framebuffer, EnumFramebufferAttachment.ColorAttachment2, scene.Motion, 0); + seam.AttachTexture(scene.Framebuffer, EnumFramebufferAttachment.ColorAttachment3, scene.Extra, 0); + Assert.True(seam.CheckFramebufferComplete(scene.Framebuffer, out string status), status); + return scene; + } + + private static void BaseState(VulkanDevice seam) + { + seam.SetViewport(0, 0, Size, Size); + seam.SetScissorEnabled(false); + seam.SetDepthTest(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.SetColorMask(true, true, true, true); + } + + /// A frame that clears every attachment to a known value, with every draw buffer selected. + private static void SeedFrame(VulkanDevice seam, Scene scene) + { + seam.BeginFrame(); + seam.BindFramebuffer(scene.Framebuffer); + BaseState(seam); + seam.SetDrawBuffers(scene.Framebuffer, 0b1111); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearColor(1, 0.2f, 0.4f, 0.6f, 1f); + seam.ClearColor(2, 0.125f, 0.25f, 0.375f, 0.5f); + seam.ClearColor(3, 0.4f, 0.6f, 0.8f, 1f); + seam.Present(); + } + + private static void AssertEveryPixel(byte[] texels, int bytesPerPixel, byte[] expected, string what) + { + Assert.Equal(Size * Size * bytesPerPixel, texels.Length); + for (int p = 0; p < texels.Length; p += bytesPerPixel) + { + for (int c = 0; c < bytesPerPixel; c++) + { + if (texels[p + c] != expected[c]) + { + Assert.Fail($"{what}: pixel {p / bytesPerPixel} byte {c} is {texels[p + c]}, expected {expected[c]}"); + } + } + } + } + + private static byte[] Half(float r, float g, float b, float a) + { + var bytes = new byte[8]; + BitConverter.TryWriteBytes(bytes.AsSpan(0, 2), (Half)r); + BitConverter.TryWriteBytes(bytes.AsSpan(2, 2), (Half)g); + BitConverter.TryWriteBytes(bytes.AsSpan(4, 2), (Half)b); + BitConverter.TryWriteBytes(bytes.AsSpan(6, 2), (Half)a); + return bytes; + } + + /// + /// A pass over Primary with the default colour set, a motion window (all + /// three), a motion-only window, a restore, a masked-out clear and an + /// all-false colour-mask clear. The fourth slot never has its draw buffer on + /// in the pass, though every program writes it: bit-identical to the previous + /// frame's seed. The motion-only draw writes exactly the motion attachment. + /// All of it in one rendering scope; no restart from a mask change. + /// + [SkippableTheory] + [MemberData(nameof(Tiers))] + public void MotionWindowsAreWriteMasksInsideOneScope(string tierToken) + { + ColorWriteTier tier = Tier(tierToken); + Skip.IfNot(TryCreateDevice(tier, out VulkanDevice? device), "Vulkan or tier " + tier + " unavailable."); + using (device) + { + VulkanDevice seam = device!; + int opaque = GpuTest.LinkProgram(seam, FullscreenVertex, + FourOutputs("1.0, 0.0, 0.0, 1.0", "0.0, 1.0, 0.0, 1.0", "0.75, 0.75, 0.75, 0.75", "1.0, 1.0, 1.0, 1.0"), + "mw-opaque"); + int motionOnly = GpuTest.LinkProgram(seam, FullscreenVertex, + FourOutputs("0.0, 0.0, 1.0, 1.0", "0.0, 0.0, 1.0, 1.0", "1.0, 0.5, 0.0, 1.0", "0.0, 0.0, 0.0, 0.0"), + "mw-motion-only"); + Scene scene = CreateScene(seam); + SeedFrame(seam, scene); + + seam.BeginFrame(); + seam.BindFramebuffer(scene.Framebuffer); + BaseState(seam); + long scopesBefore = device!.ScopesOpenedForTests; + + // Default colour set: colour and glow; motion and extra masked. + seam.SetDrawBuffers(scene.Framebuffer, 0b0011); + seam.UseProgram(opaque); + seam.DrawFullscreenTriangle(); + + // Motion window (EnableMotionDrawBuffers), then the motion-only window. + seam.SetDrawBuffers(scene.Framebuffer, 0b0111); + seam.DrawFullscreenTriangle(); + seam.SetDrawBuffers(scene.Framebuffer, 0b0100); + seam.UseProgram(motionOnly); + seam.DrawFullscreenTriangle(); + + // Restore, then two clears that must not land: motion masked out, and + // colour through an all-false glColorMask. + seam.SetDrawBuffers(scene.Framebuffer, 0b0011); + seam.ClearColor(2, 0f, 0f, 0f, 0f); + seam.SetColorMask(false, false, false, false); + seam.ClearColor(0, 0f, 0f, 1f, 1f); + seam.SetColorMask(true, true, true, true); + seam.UseProgram(opaque); + seam.SetDrawBuffers(scene.Framebuffer, 0b0001); + seam.DrawFullscreenTriangle(); + + long scopes = device.ScopesOpenedForTests - scopesBefore; + long maskRestarts = device.MaskRestartsForTests; + long splits = device.FeedbackSplitsForTests; + + byte[] colour = device.ReadBackLevel0ForTests(scene.Colour); + byte[] glow = device.ReadBackLevel0ForTests(scene.Glow); + byte[] motion = device.ReadBackLevel0ForTests(scene.Motion); + byte[] extra = device.ReadBackLevel0ForTests(scene.Extra); + seam.Present(); + + _output.WriteLine($"tier={tier} scopes={scopes} mask_restarts={maskRestarts} feedback_splits={splits}"); + Assert.Equal(1, scopes); + Assert.Equal(0, maskRestarts); + Assert.Equal(0, splits); + + AssertEveryPixel(colour, 4, new byte[] { 255, 0, 0, 255 }, "colour"); + AssertEveryPixel(glow, 4, new byte[] { 0, 255, 0, 255 }, "glow"); + AssertEveryPixel(motion, 8, Half(1f, 0.5f, 0f, 1f), "motion (motion-only window wrote it last)"); + AssertEveryPixel(extra, 4, new byte[] { 102, 153, 204, 255 }, "extra (draw buffer never on)"); + + GpuTest.AssertClean(seam); + } + } + + /// + /// The OIT merge: standard blending on colour, additive (ONE, ONE) on the + /// motion attachment (ApplyOptimumMotionAccumulateBlendState), a program that + /// adds only to motion's blue. Red, green and alpha of the RGBA16F motion + /// texels stay bit-exact; glow, which the program does not write, is untouched. + /// + [SkippableTheory] + [MemberData(nameof(Tiers))] + public void OitMergeAdditiveOnMotionLeavesRedGreenAndAlphaBitExact(string tierToken) + { + ColorWriteTier tier = Tier(tierToken); + Skip.IfNot(TryCreateDevice(tier, out VulkanDevice? device), "Vulkan or tier " + tier + " unavailable."); + using (device) + { + VulkanDevice seam = device!; + int merge = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + layout(location = 0) out vec4 outColor; + layout(location = 2) out vec4 outMotion; + void main(void) + { + outColor = vec4(1.0, 1.0, 1.0, 0.5); + outMotion = vec4(0.0, 0.0, 0.25, 0.0); + } + """, "mw-oit-merge"); + Scene scene = CreateScene(seam); + SeedFrame(seam, scene); + + seam.BeginFrame(); + seam.BindFramebuffer(scene.Framebuffer); + BaseState(seam); + seam.SetDrawBuffers(scene.Framebuffer, 0b0111); + seam.UseProgram(merge); + // ApplyTransparentMergeBlendState, then the motion accumulate override. + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetBlendFuncSeparate(0, 770, 771, 770, 771); + seam.SetBlendEquation(2, 32774); + seam.SetBlendFuncSeparate(2, 1, 1, 1, 1); + seam.DrawFullscreenTriangle(); + + // The accumulate override must not outlive the draw: a second draw with + // replace blending on motion keeps the scope and rewrites blue only. + seam.SetBlendFuncSeparate(2, 1, 0, 1, 0); + seam.SetDrawBuffers(scene.Framebuffer, 0b0001); + seam.DrawFullscreenTriangle(); + + long maskRestarts = device!.MaskRestartsForTests; + byte[] colour = device.ReadBackLevel0ForTests(scene.Colour); + byte[] glow = device.ReadBackLevel0ForTests(scene.Glow); + byte[] motion = device.ReadBackLevel0ForTests(scene.Motion); + seam.Present(); + + Assert.Equal(0, maskRestarts); + byte[] seed = Half(0.125f, 0.25f, 0.375f, 0.5f); + byte[] merged = Half(0.125f, 0.25f, 0.625f, 0.5f); + for (int p = 0; p < motion.Length; p += 8) + { + Assert.Equal(seed.AsSpan(0, 4).ToArray(), motion.AsSpan(p, 4).ToArray()); // red, green + Assert.Equal(seed.AsSpan(6, 2).ToArray(), motion.AsSpan(p + 6, 2).ToArray()); // alpha + } + AssertEveryPixel(motion, 8, merged, "motion after additive merge"); + AssertEveryPixel(glow, 4, new byte[] { 51, 102, 153, 255 }, "glow (not written by the merge)"); + + // Colour: two draws of (1,1,1,0.5) over black with SRC_ALPHA blending. + Assert.InRange(colour[0], 189, 193); + Assert.Equal(colour[0], colour[1]); + + GpuTest.AssertClean(seam); + } + } + + /// + /// The final composition shape: draw buffers select Primary 0 only while the + /// program samples Primary 1. The sampled slot leaves the draw's pass, colour receives + /// glow's texels exactly, glow keeps them; selecting glow again lets it rejoin and + /// a write lands. Validation stays clean. The stated draw declares its pass without the + /// sampled slot before any scope opens, so nothing splits on either path. + /// + [SkippableTheory] + [MemberData(nameof(TiersWithFrameGraph))] + public void CompositionSamplesAnAttachmentItsDrawBuffersExclude(string tierToken, bool frameGraph) + { + ColorWriteTier tier = Tier(tierToken); + Skip.IfNot(TryCreateDevice(tier, out VulkanDevice? device), "Vulkan or tier " + tier + " unavailable."); + using (device) + { + VulkanDevice seam = device!; + seam.FrameGraphEnabled = frameGraph; + int compose = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D glowTex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = texture(glowTex, uv); } + """, "mw-compose"); + int writeGlow = GpuTest.LinkProgram(seam, FullscreenVertex, + FourOutputs("0.0, 0.0, 0.0, 1.0", "1.0, 0.0, 0.0, 1.0", "0.0, 0.0, 0.0, 0.0", "0.0, 0.0, 0.0, 0.0"), + "mw-write-glow"); + Scene scene = CreateScene(seam); + SeedFrame(seam, scene); + + seam.BeginFrame(); + seam.BindFramebuffer(scene.Framebuffer); + BaseState(seam); + seam.SetDrawBuffers(scene.Framebuffer, 0b0001); + seam.ClearColor(0, 0f, 0f, 0f, 1f); // inference: opens the scope with glow in it, masked + seam.UseProgram(compose); + seam.SetSamplerUnit(compose, "glowTex", 0); + seam.BindTexture(0, scene.Glow); + seam.DrawFullscreenTriangle(); + byte[] composed = device!.ReadBackLevel0ForTests(scene.Colour); + byte[] glowAfterCompose = device.ReadBackLevel0ForTests(scene.Glow); + long splitsAfterCompose = device.FeedbackSplitsForTests; + + seam.BindTexture(0, 0); + seam.BindFramebuffer(scene.Framebuffer); + seam.SetDrawBuffers(scene.Framebuffer, 0b0010); + seam.UseProgram(writeGlow); + seam.DrawFullscreenTriangle(); + long maskRestarts = device.MaskRestartsForTests; + byte[] glowAfterWrite = device.ReadBackLevel0ForTests(scene.Glow); + seam.Present(); + + _output.WriteLine($"tier={tier} frameGraph={frameGraph} splits_after_compose={splitsAfterCompose} mask_restarts={maskRestarts}"); + Assert.Equal(0, splitsAfterCompose); + Assert.Equal(0, maskRestarts); + AssertEveryPixel(composed, 4, new byte[] { 51, 102, 153, 255 }, "colour = sampled glow"); + AssertEveryPixel(glowAfterCompose, 4, new byte[] { 51, 102, 153, 255 }, "glow untouched by composition"); + AssertEveryPixel(glowAfterWrite, 4, new byte[] { 255, 0, 0, 255 }, "glow written after it rejoined"); + + GpuTest.AssertClean(seam); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/MotionWriterTests.cs b/Optimum.Render.Vulkan.Tests/MotionWriterTests.cs new file mode 100644 index 00000000..3eff072d --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/MotionWriterTests.cs @@ -0,0 +1,1530 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Xunit; +using Xunit.Abstractions; +using static Optimum.Render.Vulkan.Tests.MotionFixture; + +namespace Optimum.Render.Vulkan.Tests; + +/// GPU contract for the terrain shaders' TAA motion attachment. +public sealed class TerrainMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + + [SkippableTheory] + [InlineData("chunkopaque")] + [InlineData("chunktopsoil")] + public void TerrainWritesPreviousMinusCurrentPixelsAndDepth(string shader) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!, shader); + + // Identity projections map one NDC unit to half the render width. + // Both axes and both signs matter: a magnitude-only check misses + // flipped or swapped motion vectors. + Check(scene.Draw(0, 0, 0), 0, 0); + Check(scene.Draw(0.25f, 0, 0), 8, 0); + Check(scene.Draw(0, -0.125f, 0), 0, -4); + Check(scene.Draw(-0.1875f, 0.0625f, 0), -6, 2); + + GpuTest.AssertClean(device!); + } + } + + [SkippableTheory] + [InlineData("chunkopaque")] + [InlineData("chunktopsoil")] + public void PreviousWarpMovesTheVectorWithoutWritingUncoveredPixels(string shader) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!, shader); + float[] motion = scene.Draw(0, 0, 8); + + // Vertexwarp's phase is zero for this face. Compute its displacement + // independently of the shader output, then convert NDC to pixels. + double warp = (Math.Sin(0) + Math.Sin(0.5) + Math.Sin(1) / 3) / 30 * 8; + float expectedX = (float)(warp * Size / 2); + Assert.True(expectedX > 1); + Check(motion, expectedX, 0); + Assert.InRange(Pixel(motion, 2, 2, 3), 0, 0.001f); + + GpuTest.AssertClean(device!); + } + } + + private static void Check(float[] pixels, float expectedX, float expectedY) + { + Assert.InRange(Pixel(pixels, Size / 2, Size / 2, 0), expectedX - 0.05f, expectedX + 0.05f); + Assert.InRange(Pixel(pixels, Size / 2, Size / 2, 1), expectedY - 0.05f, expectedY + 0.05f); + Assert.InRange(Pixel(pixels, Size / 2, Size / 2, 3), 0.49f, 0.51f); + } + + private static float Pixel(float[] pixels, int x, int y, int channel) => + pixels[(y * Size + x) * 4 + channel]; + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly int framebuffer; + private readonly int motion; + private readonly int mesh; + + public Scene(VulkanDevice device, string shader) + { + this.device = device; + var variant = ShaderCorpus.Variants().Single(v => v.Name == "taa-no-ssao"); + Assert.Equal(1, variant.TaaMotion); + Assert.Equal(2, variant.TaaMotionLocation); + var stages = ShaderCorpus.BuildProgram(shader, ShaderCorpus.LoadShaderFiles(), + ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, shader, oit: true); + Assert.True(device.GetUniformLocation(program, "taaRenderSize") >= 0, + shader + " has no motion writer"); + + int unit = BindEveryDeclaredSampler(device, device, program); + int atlas = CreateWhiteTexture(device); + foreach (string sampler in new[] { "terrainTex", "terrainTexLinear" }) + { + device.SetSamplerUnit(program, sampler, unit); + device.BindTexture(unit++, atlas); + } + + MotionTarget target = CreateMotionTarget(device, Size); + framebuffer = target.Framebuffer; + motion = target.MotionTexture; + mesh = CreateFaceMesh(device); + } + + public float[] Draw(float cameraX, float cameraY, float previousWarp) + { + device.BeginFrame(); + device.BindFramebuffer(framebuffer); + device.ClearColor(0, 0, 0, 0, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.UseProgram(program); + + SetMatrix(device, program, "projectionMatrix", Identity); + SetMatrix(device, program, "modelViewMatrix", Identity); + SetMatrix(device, program, "prevProjectionMatrix", Identity); + SetMatrix(device, program, "prevModelViewMatrix", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixFar", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixNear", Identity); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", 0, 0); + SetViewUniforms(); + SetWarpUniforms(previousWarp); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMesh(mesh); + + float[] pixels = ReadMotion(device, motion, Size); + device.Present(); + return pixels; + } + + private void SetViewUniforms() + { + SetFloat(device, program, "viewDistance", 1024); + SetFloat(device, program, "viewDistanceLod0", 1024); + SetFloat(device, program, "alphaTest", 0.001f); + SetFloat(device, program, "zNear", 0.1f); + SetFloat(device, program, "zFar", 1024); + SetFloat(device, program, "shadowRangeFar", 1024); + SetFloat(device, program, "shadowRangeNear", 64); + SetFloat(device, program, "shadowMapWidthInv", 1); + SetFloat(device, program, "shadowMapHeightInv", 1); + SetFloat2(device, program, "blockTextureSize", 1, 1); + SetFloat3(device, program, "rgbaAmbientIn", 1, 1, 1); + SetFloat2(device, program, "frameSize", Size, Size); + } + + private void SetWarpUniforms(float previousWarp) + { + SetInt(device, program, "perceptionEffectId", 1); + SetInt(device, program, "prevPerceptionEffectId", 1); + SetFloat(device, program, "windWaveIntensity", 1); + SetFloat(device, program, "waterWaveIntensity", 1); + SetFloat(device, program, "prevWindWaveIntensity", 1); + SetFloat(device, program, "prevWaterWaveIntensity", 1); + SetFloat(device, program, "globalWarpIntensity", 0); + SetFloat(device, program, "prevGlobalWarpIntensity", previousWarp); + } + + } +} + +/// Direct RGBA16F contract for the depth-tested sky motion pass. +public sealed class SkyMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private const float Yaw = 0.1f; + + private static float[] InverseJittered(float x, float y) => + [ + 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, -1, + -2 * x / Size, -2 * y / Size, -1, 1, + ]; + + private static float[] PreviousViewProjection(float yaw) + { + float c = MathF.Cos(yaw), s = MathF.Sin(yaw); + return [c, 0, s, s, 0, 1, 0, 0, s, 0, -c, -c, 0, 0, -1, 0]; + } + + private static (float X, float Y) ExpectedMotion(int x, int y, float yaw, float jx, float jy) + { + float fragX = x + 0.5f, fragY = y + 0.5f; + float nx = fragX / Size * 2 - 1 - 2 * jx / Size; + float ny = fragY / Size * 2 - 1 - 2 * jy / Size; + float c = MathF.Cos(yaw), s = MathF.Sin(yaw); + float denominator = s * nx + c; + float previousX = ((c * nx - s) / denominator * 0.5f + 0.5f) * Size; + float previousY = (ny / denominator * 0.5f + 0.5f) * Size; + return (previousX - (fragX - jx), previousY - (fragY - jy)); + } + + [SkippableTheory] + [InlineData(1f, 0f, 0f)] + [InlineData(0f, 0f, 0f)] + [InlineData(0.5f, 0f, 0f)] + [InlineData(1f, 0.375f, -0.25f)] + [InlineData(1f, -0.5f, 0.5f)] + public void SkyRotationAndCloudCoverageWriteOnlyMotion(float coverage, float jx, float jy) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + (float[] motion, byte[] color) = scene.Draw(coverage, jx, jy); + foreach ((int x, int y) in new[] { (32, 32), (48, 40), (20, 12), (56, 56) }) + { + int i = (y * Size + x) * 4; + (float expectedX, float expectedY) = ExpectedMotion(x, y, Yaw, jx, jy); + Assert.InRange(motion[i], expectedX - 0.05f, expectedX + 0.05f); + Assert.InRange(motion[i + 1], expectedY - 0.05f, expectedY + 0.05f); + Assert.InRange(motion[i + 2], coverage + coverage * (1 - coverage) - 0.01f, + coverage + coverage * (1 - coverage) + 0.01f); + Assert.InRange(motion[i + 3], 0.99f, 1.01f); + } + Assert.True(Math.Abs(motion[(32 * Size + 32) * 4]) > 1); + byte[] expectedColor = [51, 102, 153, 255]; + for (int i = 0; i < color.Length; i++) + Assert.InRange(color[i], expectedColor[i % 4] - 1, expectedColor[i % 4] + 1); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void SkyDepthTestPreservesEarlierSurfaceWriter() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + (float[] motion, _) = scene.Draw(1f, seedLeftHalf: true); + int covered = (32 * Size + 12) * 4; + int sky = (32 * Size + 52) * 4; + Assert.InRange(motion[covered], 7.99f, 8.01f); + Assert.InRange(motion[covered + 1], -8.01f, -7.99f); + Assert.InRange(motion[covered + 2], 0.24f, 0.26f); + Assert.InRange(motion[covered + 3], 0.49f, 0.51f); + Assert.InRange(motion[sky + 2], 0.99f, 1.01f); + Assert.InRange(motion[sky + 3], 0.99f, 1.01f); + GpuTest.AssertClean(device!); + } + } + + [SkippableTheory] + [InlineData(1.7, 0.0)] + [InlineData(1.7, -0.35)] + [InlineData(25.0, 0.2)] + public void StationaryElevatedCameraHasZeroSkyMotion(double height, double pitch) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + const double near = 0.0689, far = 60.0, fov = 70.0 * Math.PI / 180.0; + double[] projection = Mat4d.Perspective(Mat4d.Create(), fov, 1.0, near, far); + double[] view = Mat4d.Identity(Mat4d.Create()); + view = Mat4d.RotateX(view, view, pitch); + view = Mat4d.Translate(view, view, 0, -height, 0); + double[] vp = Mat4d.Mul(Mat4d.Create(), projection, view); + double[] inverse = Mat4d.Invert(Mat4d.Create(), vp); + Assert.NotNull(inverse); + Assert.True(height / far * (Size / 2.0) / Math.Tan(fov / 2.0) > 0.6); + using (device) + { + var scene = new Scene(device!); + (float[] motion, _) = scene.Draw(0f, + inverse: Array.ConvertAll(inverse!, v => (float)v), + previous: Array.ConvertAll(vp, v => (float)v)); + foreach ((int x, int y) in new[] { (32, 32), (8, 56), (56, 8), (20, 44) }) + { + int i = (y * Size + x) * 4; + Assert.InRange(motion[i], -0.05f, 0.05f); + Assert.InRange(motion[i + 1], -0.05f, 0.05f); + Assert.InRange(motion[i + 3], 0.99f, 1.01f); + } + GpuTest.AssertClean(device!); + } + } + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly MotionTarget target; + private readonly int program; + private readonly int reveal; + private readonly int seedProgram; + private readonly int seedMesh; + + internal Scene(VulkanDevice device) + { + this.device = device; + ShaderCorpus.ShaderVariant variant = ShaderCorpus.Variants().First(v => v.Name == "taa-no-ssao"); + Assert.Equal(1, variant.TaaMotion); + Assert.Equal(2, variant.TaaMotionLocation); + program = LinkFromCorpus(device, ShaderCorpus.BuildProgram("taa-skymotion", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant), "taa-skymotion", oit: true); + Assert.True(device.GetUniformLocation(program, "taaRenderSize") >= 0); + target = CreateMotionTarget(device, Size); + reveal = device.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + device.SetTextureParameter(reveal, OptimumGlConstants.TextureMinFilter, 9728); + device.SetTextureParameter(reveal, OptimumGlConstants.TextureMagFilter, 9728); + int revealTarget = device.CreateFramebuffer(Size, Size); + device.AttachTexture(revealTarget, EnumFramebufferAttachment.ColorAttachment0, reveal, 0); + device.SetDrawBuffers(revealTarget, 0b1); + RevealTarget = revealTarget; + const string vertex = "#version 330 core\nlayout(location=0) in vec3 xyz; void main() { gl_Position=vec4(xyz,1); }"; + const string fragment = "#version 330 core\nlayout(location=2) out vec4 outMotion; void main() { outMotion=vec4(8,-8,0.25,gl_FragCoord.z); }"; + seedProgram = LinkFromCorpus(device, + [ + new() { Stage = EnumShaderType.VertexShader, Code = vertex, PrefixCode = "", Filename = "sky-seed.vsh" }, + new() { Stage = EnumShaderType.FragmentShader, Code = fragment, PrefixCode = "", Filename = "sky-seed.fsh" }, + ], "sky-seed", oit: true); + MeshData left = new(4, 6, withNormals: false, withUv: false, withRgba: false, withFlags: false) + { + xyz = [-1, -1, 0, 0, -1, 0, 0, 1, 0, -1, 1, 0], + VerticesCount = 4, + Indices = [0, 1, 2, 0, 2, 3], + IndicesCount = 6, + }; + seedMesh = device.CreateMesh(left, staticDraw: true); + Assert.True(seedMesh > 0, device.GetError() ?? "sky seed upload failed"); + } + + private int RevealTarget { get; } + + internal (float[] Motion, byte[] Color) Draw(float coverage, float jx = 0, float jy = 0, + bool seedLeftHalf = false, float[]? inverse = null, float[]? previous = null) + { + device.BeginFrame(); + device.BindFramebuffer(RevealTarget); + float revealValue = 1 - coverage; + device.ClearColor(0, revealValue, revealValue, revealValue, 1); + device.BindFramebuffer(target.Framebuffer); + device.SetDrawBuffers(target.Framebuffer, 0b111); + device.ClearColor(0, 0.2f, 0.4f, 0.6f, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.SetViewport(0, 0, Size, Size); + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.SetDepthFunc(0x203); + device.SetDrawBuffers(target.Framebuffer, 1 << 2); + if (seedLeftHalf) + { + device.SetDepthTest(true); + device.SetDepthMask(true); + device.UseProgram(seedProgram); + device.DrawMesh(seedMesh); + } + device.UseProgram(program); + device.SetSamplerUnit(program, "transparentRevealTex", 14); + device.BindTexture(14, reveal); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", jx, jy); + SetMatrix(device, program, "taaInvViewProjJittered", inverse ?? InverseJittered(jx, jy)); + SetMatrix(device, program, "taaPrevViewProj", previous ?? PreviousViewProjection(Yaw)); + SetFloat(device, program, "taaCloudReactive", 1); + device.SetDepthTest(true); + device.SetDepthMask(false); + device.DrawFullscreenTriangle(); + float[] motion = ReadMotion(device, target.MotionTexture, Size); + byte[] color = device.ReadBackLevel0ForTests(target.ColorTexture); + device.Present(); + return (motion, color); + } + } +} + +/// Previous-object motion written by the actual standard shader. +public sealed class StandardMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + + private static float[] Translation(float x, float y, float z) => + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + x, y, z, 1, + ]; + + // At view z = -1 the shear moves the raster position by exactly the + // requested subpixel jitter; the previous projection remains unjittered. + private static float[] Projection(float jitterX, float jitterY) + { + var projection = new float[16]; + projection[0] = projection[5] = 1; + projection[8] = -2 * jitterX / Size; + projection[9] = -2 * jitterY / Size; + projection[10] = projection[11] = projection[14] = -1; + return projection; + } + + [SkippableFact] + public void PreviousModelAndMissingHistoryProduceIndependentPixelVectors() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + Check(scene.Draw(Identity), 0, 0, 0.5f, 0); + Check(scene.Draw(Translation(0.25f, 0, 0)), 8, 0, 0.5f, 0); + Check(scene.Draw(Translation(0, -0.125f, 0)), 0, -4, 0.5f, 0); + Check(scene.Draw(Translation(-0.1875f, 0.0625f, 0)), -6, 2, 0.5f, 0); + + // A stale object transform must not leak into a draw with no history. + Check(scene.Draw(Translation(-0.5f, 0.5f, 0), history: false, + cameraX: 0.25f, cameraY: -0.125f), 8, -4, 0.5f, 1); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void WarpOptOutAndBehindCameraKeepTheirMotionContract() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + double offset = (Math.Sin(0) + Math.Sin(0.5) + Math.Sin(1) / 3) / 30 * 8; + float expected = (float)(offset * Size / 2); + Assert.True(expected > 1); + Check(scene.Draw(Identity, previousWarp: 8), expected, 0, 0.5f, 0); + Check(scene.Draw(Identity, previousWarp: 8, noWarp: true), 0, 0, 0.5f, 0); + + float[] behindProjection = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, -1, + 0, 0, 0, 0, + ]; + Check(scene.Draw(Identity, previousView: Translation(0, 0, 1), + previousProjection: behindProjection, reactive: 0.6f), + 0, 0, 0, 0.6f); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void MoverMotionExcludesProjectionJitterAndIgnoresStaleHistory() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!, perspective: true); + Check(scene.Draw(Identity, noWarp: true, jitterX: 0.37f, jitterY: -0.24f), + 0, 0, 0.5f, 0); + Check(scene.Draw(Identity, noWarp: true, jitterX: -0.5f, jitterY: 0.5f), + 0, 0, 0.5f, 0); + Check(scene.Draw(Translation(0.25f, 0, 0), noWarp: true, + jitterX: 0.5f, jitterY: -0.5f), 8, 0, 0.5f, 0); + Check(scene.Draw(Translation(0, -0.125f, 0), noWarp: true, + jitterX: -0.5f, jitterY: 0.5f), 0, -4, 0.5f, 0); + + float[] previous = Translation(-0.1875f, 0.0625f, 0); + float[] unjittered = scene.Draw(previous, noWarp: true); + float[] jittered = scene.Draw(previous, noWarp: true, + jitterX: 0.31f, jitterY: 0.47f); + Check(jittered, -6, 2, 0.5f, 0); + int centre = ((Size / 2) * Size + Size / 2) * 4; + Assert.InRange(jittered[centre], unjittered[centre] - 0.05f, + unjittered[centre] + 0.05f); + Assert.InRange(jittered[centre + 1], unjittered[centre + 1] - 0.05f, + unjittered[centre + 1] + 0.05f); + + Check(scene.Draw(Translation(-0.5f, 0.5f, 0), history: false, + noWarp: true, cameraX: 0.25f, cameraY: -0.125f, + jitterX: 0.42f, jitterY: 0.13f), 8, -4, 0.5f, 1); + GpuTest.AssertClean(device!); + } + } + + private static void Check(float[] pixels, float x, float y, float depth, float reactive) + { + int centre = ((Size / 2) * Size + Size / 2) * 4; + Assert.InRange(pixels[centre], x - 0.05f, x + 0.05f); + Assert.InRange(pixels[centre + 1], y - 0.05f, y + 0.05f); + Assert.InRange(pixels[centre + 2], reactive - 0.01f, reactive + 0.01f); + Assert.InRange(pixels[centre + 3], depth - 0.01f, depth + 0.01f); + } + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly MotionTarget target; + private readonly int mesh; + private readonly bool perspective; + + public Scene(VulkanDevice device, bool perspective = false) + { + this.device = device; + this.perspective = perspective; + var variant = new ShaderCorpus.ShaderVariant + { + Name = "taa-standard", + TaaMotion = 1, + TaaMotionLocation = 2, + }; + var stages = ShaderCorpus.BuildProgram("standard", ShaderCorpus.LoadShaderFiles(), + ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "standard", oit: false); + foreach (string required in new[] { "taaRenderSize", "taaHistoryValid", "prevModelMatrix" }) + Assert.True(device.GetUniformLocation(program, required) >= 0, required + " is missing"); + BindEveryDeclaredSampler(device, device, program); + target = CreateMotionTarget(device, Size); + mesh = CreateFaceMesh(device, perspective ? -1 : 0); + } + + public float[] Draw(float[] previousModel, bool history = true, bool noWarp = false, + float cameraX = 0, float cameraY = 0, float previousWarp = 0, + float[]? previousView = null, float[]? previousProjection = null, float reactive = 0, + float jitterX = 0, float jitterY = 0) + { + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.ClearColor(0, 0, 0, 0, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.UseProgram(program); + + SetMatrix(device, program, "projectionMatrix", + perspective ? Projection(jitterX, jitterY) : Identity); + SetMatrix(device, program, "viewMatrix", Identity); + SetMatrix(device, program, "modelMatrix", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixFar", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixNear", Identity); + SetMatrix(device, program, "prevProjectionMatrix", + previousProjection ?? (perspective ? Projection(0, 0) : Identity)); + SetMatrix(device, program, "prevViewMatrix", previousView ?? Identity); + SetMatrix(device, program, "prevModelMatrix", previousModel); + SetInt(device, program, "taaHistoryValid", history ? 1 : 0); + SetFloat(device, program, "taaReactive", history ? reactive : 1); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", jitterX, jitterY); + SetInt(device, program, "dontWarpVertices", noWarp ? 1 : 0); + SetFloat(device, program, "alphaTest", -1); + SetFloat(device, program, "viewDistance", 1024); + SetFloat(device, program, "viewDistanceLod0", 1024); + SetFloat(device, program, "zNear", 0.1f); + SetFloat(device, program, "zFar", 1024); + SetFloat(device, program, "shadowRangeFar", 1024); + SetFloat(device, program, "shadowRangeNear", 64); + SetFloat(device, program, "shadowMapWidthInv", 1); + SetFloat(device, program, "shadowMapHeightInv", 1); + SetFloat3(device, program, "rgbaAmbientIn", 1, 1, 1); + SetFloat4(device, program, "rgbaLightIn", 1, 1, 1, 1); + SetFloat4(device, program, "rgbaFogIn", 1, 1, 1, 1); + SetFloat4(device, program, "rgbaTint", 1, 1, 1, 1); + SetFloat4(device, program, "averageColor", 1, 1, 1, 1); + SetFloat2(device, program, "frameSize", Size, Size); + SetInt(device, program, "perceptionEffectId", 1); + SetInt(device, program, "prevPerceptionEffectId", 1); + SetFloat(device, program, "windWaveIntensity", 1); + SetFloat(device, program, "waterWaveIntensity", 1); + SetFloat(device, program, "prevWindWaveIntensity", 1); + SetFloat(device, program, "prevWaterWaveIntensity", 1); + SetFloat(device, program, "globalWarpIntensity", 0); + SetFloat(device, program, "prevGlobalWarpIntensity", previousWarp); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMesh(mesh); + float[] pixels = ReadMotion(device, target.MotionTexture, Size); + device.Present(); + return pixels; + } + } +} + +/// The instanced shader must read each draw instance's own history. +public sealed class InstancedMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + + private static float[] Translation(float x, float y, float z) => + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + x, y, z, 1, + ]; + + private readonly record struct Instance(float[] Current, float[] Previous, + bool History = true, float Reactive = 0); + + [SkippableFact] + public void OneDrawKeepsPreviousTransformPerInstance() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + Check(scene.Draw([new Instance(Identity, Identity)]), 32, 0, 0, 0.5f, 0); + Check(scene.Draw([new Instance(Identity, Translation(0.25f, 0, 0))]), + 32, 8, 0, 0.5f, 0); + Check(scene.Draw([new Instance(Identity, Translation(-0.1875f, 0.0625f, 0))]), + 32, -6, 2, 0.5f, 0); + + // Two transforms in one draw prove the previous matrix is sourced + // from the instance stream rather than a shared uniform. + float[] pixels = scene.Draw( + [ + new Instance(Translation(-0.5f, 0, 0), Translation(-0.25f, 0, 0)), + new Instance(Translation(0.5f, 0, 0), Translation(0.5f, -0.125f, 0)), + ]); + Check(pixels, 16, 8, 0, 0.5f, 0); + Check(pixels, 48, 0, -4, 0.5f, 0); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void MissingAndBehindCameraHistoryKeepReactiveInformation() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + Check(scene.Draw([new Instance(Identity, Translation(-0.5f, 0.5f, 0), + History: false, Reactive: 1)], cameraX: 0.25f, cameraY: -0.125f), + 32, 8, -4, 0.5f, 1); + + float[] behindProjection = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, -1, + 0, 0, 0, 0, + ]; + Check(scene.Draw([new Instance(Identity, Identity, Reactive: 0.6f)], + previousView: Translation(0, 0, 1), previousProjection: behindProjection), + 32, 0, 0, 0, 0.6f); + GpuTest.AssertClean(device!); + } + } + + private static void Check(float[] pixels, int x, float expectedX, float expectedY, + float depth, float reactive) + { + int pixel = ((Size / 2) * Size + x) * 4; + Assert.InRange(pixels[pixel], expectedX - 0.05f, expectedX + 0.05f); + Assert.InRange(pixels[pixel + 1], expectedY - 0.05f, expectedY + 0.05f); + Assert.InRange(pixels[pixel + 2], reactive - 0.01f, reactive + 0.01f); + Assert.InRange(pixels[pixel + 3], depth - 0.01f, depth + 0.01f); + } + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly MotionTarget target; + + public Scene(VulkanDevice device) + { + this.device = device; + var variant = new ShaderCorpus.ShaderVariant + { + Name = "taa-instanced", + TaaMotion = 1, + TaaMotionLocation = 2, + }; + var stages = ShaderCorpus.BuildProgram("instanced", ShaderCorpus.LoadShaderFiles(), + ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "instanced", oit: false); + foreach (string required in new[] { "taaRenderSize", "prevModelViewMatrix" }) + Assert.True(device.GetUniformLocation(program, required) >= 0, required + " is missing"); + BindEveryDeclaredSampler(device, device, program); + target = CreateMotionTarget(device, Size); + } + + public float[] Draw(Instance[] instances, float cameraX = 0, float cameraY = 0, + float[]? previousView = null, float[]? previousProjection = null) + { + int mesh = device.CreateMesh(BuildMesh(instances), staticDraw: false); + Assert.True(mesh > 0, device.GetError() ?? "instanced face upload failed"); + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.ClearColor(0, 0, 0, 0, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.UseProgram(program); + SetMatrix(device, program, "projectionMatrix", Identity); + SetMatrix(device, program, "modelViewMatrix", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixFar", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixNear", Identity); + SetMatrix(device, program, "prevProjectionMatrix", previousProjection ?? Identity); + SetMatrix(device, program, "prevModelViewMatrix", previousView ?? Identity); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", 0, 0); + SetFloat(device, program, "alphaTest", -1); + SetFloat(device, program, "viewDistance", 1024); + SetFloat(device, program, "viewDistanceLod0", 1024); + SetFloat(device, program, "zNear", 0.1f); + SetFloat(device, program, "zFar", 1024); + SetFloat(device, program, "shadowRangeFar", 1024); + SetFloat(device, program, "shadowRangeNear", 64); + SetFloat(device, program, "shadowMapWidthInv", 1); + SetFloat(device, program, "shadowMapHeightInv", 1); + SetFloat3(device, program, "rgbaAmbientIn", 1, 1, 1); + SetFloat4(device, program, "rgbaFogIn", 1, 1, 1, 1); + SetFloat4(device, program, "averageColor", 1, 1, 1, 1); + SetFloat2(device, program, "frameSize", Size, Size); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMeshInstanced(mesh, instances.Length); + float[] pixels = ReadMotion(device, target.MotionTexture, Size); + device.Present(); + return pixels; + } + + private static MeshData BuildMesh(Instance[] instances) + { + MeshData mesh = CreateFaceData(); + CustomMeshDataPartFloat stream = OptimumInstanceMotion.CreateInstanceFloats(instances.Length); + for (int i = 0; i < instances.Length; i++) + { + int start = i * OptimumInstanceMotion.InstanceFloats; + for (int channel = 0; channel < 4; channel++) + stream.Values[start + OptimumInstanceMotion.LightOffset + channel] = 1; + Array.Copy(instances[i].Current, 0, stream.Values, + start + OptimumInstanceMotion.TransformOffset, 16); + Array.Copy(instances[i].Previous, 0, stream.Values, + start + OptimumInstanceMotion.PrevTransformOffset, 16); + stream.Values[start + OptimumInstanceMotion.MetaOffset] = instances[i].History ? 1 : 0; + stream.Values[start + OptimumInstanceMotion.MetaOffset + 1] = instances[i].Reactive; + } + stream.Count = instances.Length * OptimumInstanceMotion.InstanceFloats; + mesh.CustomFloats = stream; + return mesh; + } + } +} + +/// Previous skinning and model motion from the opaque entity shader. +public sealed class EntityMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private const int AnimationUboBytes = 35 * 16 * sizeof(float); + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + + private static float[] Translation(float x, float y, float z) => + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + x, y, z, 1, + ]; + + [SkippableFact] + public void SeparatePreviousBoneAndModelSourcesProduceExpectedPixels() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + Check(new Scene(device!, Identity).Draw(Identity), 0, 0, 0.5f, 0); + Check(new Scene(device!, Translation(0.25f, 0, 0)).Draw(Identity), + 8, 0, 0.5f, 0); + Check(new Scene(device!, Translation(0, -0.125f, 0)).Draw(Identity), + 0, -4, 0.5f, 0); + Check(new Scene(device!, Translation(-0.1875f, 0.0625f, 0)).Draw(Identity), + -6, 2, 0.5f, 0); + Check(new Scene(device!, Identity).Draw(Translation(-0.25f, 0.125f, 0)), + -8, 4, 0.5f, 0); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void InvalidHistoryWarpAndBehindCameraKeepTheirReactiveContract() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + Check(new Scene(device!, Translation(0.75f, 0.75f, 0)).Draw( + Translation(-0.5f, 0.5f, 0), history: false, + cameraX: 0.25f, cameraY: -0.125f), 8, -4, 0.5f, 1); + + double offset = (Math.Sin(0) + Math.Sin(0.5) + Math.Sin(1) / 3) / 30 * 8; + float expectedWarp = (float)(offset * Size / 2); + Assert.True(expectedWarp > 1); + Check(new Scene(device!, Identity).Draw(Identity, previousWarp: 8), + expectedWarp, 0, 0.5f, 0); + + float[] behindProjection = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, -1, + 0, 0, 0, 0, + ]; + Check(new Scene(device!, Identity).Draw(Identity, + previousView: Translation(0, 0, 1), + previousProjection: behindProjection, reactive: 0.6f), + 0, 0, 0, 0.6f); + GpuTest.AssertClean(device!); + } + } + + private static void Check(float[] pixels, float x, float y, float depth, float reactive) + { + int centre = ((Size / 2) * Size + Size / 2) * 4; + Assert.InRange(pixels[centre], x - 0.05f, x + 0.05f); + Assert.InRange(pixels[centre + 1], y - 0.05f, y + 0.05f); + Assert.InRange(pixels[centre + 2], reactive - 0.01f, reactive + 0.01f); + Assert.InRange(pixels[centre + 3], depth - 0.01f, depth + 0.01f); + } + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly MotionTarget target; + private readonly int mesh; + + public Scene(VulkanDevice device, float[] previousBone) + { + this.device = device; + var variant = new ShaderCorpus.ShaderVariant + { + Name = "taa-entity-opaque", + UseOit = 0, + TaaMotion = 1, + TaaMotionLocation = 2, + MaxAnimatedElements = 35, + }; + var stages = ShaderCorpus.BuildProgram("entityanimated", ShaderCorpus.LoadShaderFiles(), + ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "entityanimated", oit: false); + foreach (string required in new[] { "taaRenderSize", "taaHistoryValid" }) + Assert.True(device.GetUniformLocation(program, required) >= 0, required + " is missing"); + int unit = BindEveryDeclaredSampler(device, device, program); + int atlas = CreateWhiteTexture(device); + device.SetSamplerUnit(program, "entityTex", unit); + device.BindTexture(unit, atlas); + + int current = device.CreateUniformBuffer(program, 0, "Animation", AnimationUboBytes); + int previous = device.CreateUniformBuffer(program, 1, "AnimationPrev", AnimationUboBytes); + WriteBone(device, current, Identity); + WriteBone(device, previous, previousBone); + target = CreateMotionTarget(device, Size); + mesh = device.CreateMesh(SkinnedFace(), staticDraw: true); + Assert.True(mesh > 0, device.GetError() ?? "entity face upload failed"); + } + + public float[] Draw(float[] previousModel, bool history = true, float cameraX = 0, + float cameraY = 0, float previousWarp = 0, float[]? previousView = null, + float[]? previousProjection = null, float reactive = 0) + { + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.ClearColor(0, 0, 0, 0, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.UseProgram(program); + SetMatrix(device, program, "projectionMatrix", Identity); + SetMatrix(device, program, "viewMatrix", Identity); + SetMatrix(device, program, "modelMatrix", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixFar", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixNear", Identity); + SetMatrix(device, program, "prevProjectionMatrix", previousProjection ?? Identity); + SetMatrix(device, program, "prevViewMatrix", previousView ?? Identity); + SetMatrix(device, program, "prevModelMatrix", previousModel); + SetInt(device, program, "taaHistoryValid", history ? 1 : 0); + SetFloat(device, program, "taaReactive", history ? reactive : 1); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", 0, 0); + SetFloat(device, program, "alphaTest", -1); + SetFloat(device, program, "viewDistance", 1024); + SetFloat(device, program, "viewDistanceLod0", 1024); + SetFloat(device, program, "zNear", 0.1f); + SetFloat(device, program, "zFar", 1024); + SetFloat(device, program, "shadowRangeFar", 1024); + SetFloat(device, program, "shadowRangeNear", 64); + SetFloat(device, program, "shadowMapWidthInv", 1); + SetFloat(device, program, "shadowMapHeightInv", 1); + SetInt(device, program, "entityId", 1); + SetFloat3(device, program, "rgbaAmbientIn", 1, 1, 1); + SetFloat4(device, program, "rgbaLightIn", 1, 1, 1, 1); + SetFloat4(device, program, "rgbaFogIn", 1, 1, 1, 1); + SetFloat4(device, program, "renderColor", 1, 1, 1, 1); + SetFloat2(device, program, "frameSize", Size, Size); + SetInt(device, program, "perceptionEffectId", 1); + SetInt(device, program, "prevPerceptionEffectId", 1); + SetFloat(device, program, "windWaveIntensity", 1); + SetFloat(device, program, "waterWaveIntensity", 1); + SetFloat(device, program, "prevWindWaveIntensity", 1); + SetFloat(device, program, "prevWaterWaveIntensity", 1); + SetFloat(device, program, "globalWarpIntensity", 0); + SetFloat(device, program, "prevGlobalWarpIntensity", previousWarp); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMesh(mesh); + float[] pixels = ReadMotion(device, target.MotionTexture, Size); + device.Present(); + return pixels; + } + + private static unsafe void WriteBone(VulkanDevice device, int buffer, float[] matrix) + { + fixed (float* values = matrix) + device.UpdateUniformBuffer(buffer, (IntPtr)values, 0, 16 * sizeof(float)); + } + + private static MeshData SkinnedFace() + { + MeshData mesh = CreateFaceData(); + mesh.CustomFloats = new CustomMeshDataPartFloat(4) + { + Count = 4, + InterleaveSizes = [1], + InterleaveOffsets = [0], + InterleaveStride = 4, + }; + mesh.CustomInts = new CustomMeshDataPartInt(4) + { + Count = 4, + InterleaveSizes = [1], + InterleaveOffsets = [0], + InterleaveStride = 4, + }; + return mesh; + } + } +} + +/// The liquid velocity pass writes motion without changing shaded colour. +public sealed class LiquidMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private const float Reactive = 0.3f; + private const float LiquidClipW = 1.008f; + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + private static readonly float[] Projection = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, -1, -1, + 0, 0, -1, 0, + ]; + + private static float[] Jittered(float x, float y) + { + float[] projection = (float[])Projection.Clone(); + projection[8] -= 2 * x / Size; + projection[9] -= 2 * y / Size; + return projection; + } + + [SkippableFact] + public void CameraMotionAndProjectionJitterHaveIndependentPixelExpectations() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new Scene(device!); + Check(scene.Draw(), 0, 0, 0.5f, Reactive); + Check(scene.Draw(cameraX: 0.25f), 8 / LiquidClipW, 0, 0.5f, Reactive); + Check(scene.Draw(cameraY: -0.125f), 0, -4 / LiquidClipW, 0.5f, Reactive); + Check(scene.Draw(cameraX: -0.1875f, cameraY: 0.0625f), + -6 / LiquidClipW, 2 / LiquidClipW, 0.5f, Reactive); + Check(scene.Draw(cameraX: 0.25f, cameraY: -0.125f, + jitterX: 0.375f, jitterY: -0.25f), + 8 / LiquidClipW, -4 / LiquidClipW, 0.5f, Reactive); + Check(scene.Draw(cameraX: 0.25f, cameraY: -0.125f, + jitterX: -0.5f, jitterY: 0.5f), + 8 / LiquidClipW, -4 / LiquidClipW, 0.5f, Reactive); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void VelocityWindowPreservesEveryColorPixelAndUncoveredMotionDepth() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + Result result = new Scene(device!).Draw(cameraX: 0.25f); + Check(result, 8 / LiquidClipW, 0, 0.5f, Reactive); + Assert.InRange(Channel(result.Motion, 2, 2, 3), 0, 0.001f); + byte[] expected = [51, 102, 153, 255]; + Assert.Equal(Size * Size * 4, result.Color.Length); + for (int pixel = 0; pixel < Size * Size; pixel++) + for (int channel = 0; channel < 4; channel++) + Assert.InRange(result.Color[pixel * 4 + channel], + Math.Max(0, expected[channel] - 1), + Math.Min(255, expected[channel] + 1)); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void PreviousWaveAndBehindCameraKeepMotionAndReactiveMeaning() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + Result waves = new Scene(device!, waterFlags: 1).Draw(previousWave: 3); + float[] row = new float[24]; + int moved = 0; + for (int i = 0; i < row.Length; i++) + { + int x = i + 20; + Assert.InRange(Channel(waves.Motion, x, 32, 3), 0.48f, 0.52f); + Assert.InRange(Channel(waves.Motion, x, 32, 0), -0.05f, 0.05f); + row[i] = Channel(waves.Motion, x, 32, 1); + if (Math.Abs(row[i]) > 0.3f) moved++; + } + Assert.True(moved > row.Length / 4, "previous liquid wave wrote no vertical motion"); + Assert.True(row.Max() - row.Min() > 0.3f, "previous liquid wave lost its noise shape"); + + float[] mirrorZ = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, -1, 0, + 0, 0, 0, 1, + ]; + Check(new Scene(device!).Draw(previousView: mirrorZ), 0, 0, 0, Reactive); + GpuTest.AssertClean(device!); + } + } + + private sealed record Result(float[] Motion, byte[] Color); + + private static float Channel(float[] pixels, int x, int y, int channel) => + pixels[(y * Size + x) * 4 + channel]; + + private static void Check(Result result, float x, float y, float depth, float reactive) + { + Assert.InRange(Channel(result.Motion, 32, 32, 0), x - 0.06f, x + 0.06f); + Assert.InRange(Channel(result.Motion, 32, 32, 1), y - 0.06f, y + 0.06f); + Assert.InRange(Channel(result.Motion, 32, 32, 2), reactive - 0.01f, reactive + 0.01f); + Assert.InRange(Channel(result.Motion, 32, 32, 3), depth - 0.02f, depth + 0.02f); + } + + private sealed class Scene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly MotionTarget target; + private readonly int mesh; + + public Scene(VulkanDevice device, int waterFlags = 0) + { + this.device = device; + var variant = ShaderCorpus.Variants().Single(v => v.Name == "taa-no-ssao"); + Assert.Equal(1, variant.TaaMotion); + Assert.Equal(2, variant.TaaMotionLocation); + Assert.Equal(1, variant.WavingStuff); + var stages = ShaderCorpus.BuildProgram("chunkliquidmotion", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "chunkliquidmotion", oit: true); + Assert.True(device.GetUniformLocation(program, "taaRenderSize") >= 0); + target = CreateMotionTarget(device, Size); + mesh = device.CreateMesh(LiquidFace(waterFlags), staticDraw: true); + Assert.True(mesh > 0, device.GetError() ?? "liquid face upload failed"); + } + + public Result Draw(float cameraX = 0, float cameraY = 0, float jitterX = 0, + float jitterY = 0, float previousWave = 0, float[]? previousView = null) + { + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.SetDrawBuffers(target.Framebuffer, 0b111); + device.ClearColor(0, 0.2f, 0.4f, 0.6f, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.SetDrawBuffers(target.Framebuffer, 1 << 2); + device.UseProgram(program); + SetMatrix(device, program, "projectionMatrix", Jittered(jitterX, jitterY)); + SetMatrix(device, program, "modelViewMatrix", Identity); + SetMatrix(device, program, "prevProjectionMatrix", Projection); + SetMatrix(device, program, "prevModelViewMatrix", previousView ?? Identity); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", jitterX, jitterY); + SetFloat(device, program, "taaLiquidReactive", Reactive); + SetInt(device, program, "perceptionEffectId", 1); + SetInt(device, program, "prevPerceptionEffectId", 1); + SetFloat(device, program, "globalWarpIntensity", 0); + SetFloat(device, program, "prevGlobalWarpIntensity", 0); + SetFloat(device, program, "windWaveIntensity", 0); + SetFloat(device, program, "prevWindWaveIntensity", 0); + SetFloat(device, program, "waterWaveIntensity", 0); + SetFloat(device, program, "prevWaterWaveIntensity", previousWave); + SetFloat(device, program, "prevWaterWaveCounter", 1.234f); + SetFloat3(device, program, "origin", 0, 0, 0); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMesh(mesh); + float[] motion = ReadMotion(device, target.MotionTexture, Size); + device.SetDrawBuffers(target.Framebuffer, 0b111); + byte[] color = device.ReadBackLevel0ForTests(target.ColorTexture); + device.Present(); + return new Result(motion, color); + } + + private static MeshData LiquidFace(int waterFlags) + { + MeshData face = CreateFaceData(-1, flags: 0); + face.CustomFloats = new CustomMeshDataPartFloat + { + Values = new float[8], + Count = 8, + InterleaveOffsets = [0], + InterleaveSizes = [2], + InterleaveStride = 8, + }; + int[] flags = new int[8]; + for (int i = 0; i < 4; i++) flags[i * 2 + 1] = waterFlags; + face.CustomInts = new CustomMeshDataPartInt + { + Values = flags, + Count = 8, + InterleaveOffsets = [0, 4], + InterleaveSizes = [1, 1], + InterleaveStride = 8, + Conversion = DataConversion.Integer, + }; + return face; + } + } +} + +/// Opaque cube-particle motion and transparent-merge reactive coverage. +public sealed class ParticleMotionContractTests(ITestOutputHelper output) +{ + private const int Size = 64; + private static readonly float[] Identity = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + ]; + private static readonly float[] Projection = + [ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, -1, -1, + 0, 0, -1, 0, + ]; + + private static float[] Jittered(float x, float y) + { + float[] projection = (float[])Projection.Clone(); + projection[8] -= 2 * x / Size; + projection[9] -= 2 * y / Size; + return projection; + } + + [SkippableFact] + public void CubeParticleWritesCameraMotionReactiveAndColor() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + var scene = new CubeScene(device!); + Result still = scene.Draw(); + Check(still.Motion, 32, 0, 0, 1, 0.5f); + int centre = (32 * Size + 32) * 4; + byte[] clear = [51, 102, 153, 255]; + Assert.True(Enumerable.Range(0, 4).Any(c => + Math.Abs(still.Color[centre + c] - clear[c]) > 1), + "cube particle did not shade color attachment 0"); + + Check(scene.Draw(cameraX: 0.25f).Motion, 32, 8, 0, 1, 0.5f); + Check(scene.Draw(cameraY: -0.125f).Motion, 32, 0, -4, 1, 0.5f); + Check(scene.Draw(cameraX: -0.1875f, cameraY: 0.0625f).Motion, + 32, -6, 2, 1, 0.5f); + Check(scene.Draw(cameraX: 0.25f, cameraY: -0.125f, + jitterX: 0.375f, jitterY: -0.25f).Motion, 32, 8, -4, 1, 0.5f); + Check(scene.Draw(cameraX: 0.25f, cameraY: -0.125f, + jitterX: -0.5f, jitterY: 0.5f).Motion, 32, 8, -4, 1, 0.5f); + Check(scene.Draw().Motion, 2, 0, 0, 0, 0); + GpuTest.AssertClean(device!); + } + } + + [SkippableFact] + public void TransparentMergeAddsCoverageWithoutReplacingOpaqueVectorOrDepth() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + foreach ((float revealage, float reactive) in new[] + { + (1f, 0f), (0.25f, 0.75f), (0f, 1f), + }) + Check(new MergeScene(device!).Draw(revealage, 0), + 32, 4, -8, reactive, 0.5f); + + Check(new MergeScene(device!).Draw(1, 1), 32, 4, -8, 1, 0.5f); + GpuTest.AssertClean(device!); + } + } + + private sealed record Result(float[] Motion, byte[] Color); + + private static void Check(float[] pixels, int x, float motionX, float motionY, + float reactive, float depth) + { + int pixel = (32 * Size + x) * 4; + Assert.InRange(pixels[pixel], motionX - 0.06f, motionX + 0.06f); + Assert.InRange(pixels[pixel + 1], motionY - 0.06f, motionY + 0.06f); + Assert.InRange(pixels[pixel + 2], reactive - 0.01f, reactive + 0.01f); + Assert.InRange(pixels[pixel + 3], depth - 0.02f, depth + 0.02f); + } + + private sealed class CubeScene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly MotionTarget target; + private readonly int mesh; + + public CubeScene(VulkanDevice device) + { + this.device = device; + var variant = ShaderCorpus.Variants().Single(v => v.Name == "taa-no-ssao"); + var stages = ShaderCorpus.BuildProgram("particlescube", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "particlescube", oit: true); + Assert.True(device.GetUniformLocation(program, "taaRenderSize") >= 0); + target = CreateMotionTarget(device, Size); + mesh = device.CreateMesh(CubeFace(), staticDraw: true); + Assert.True(mesh > 0, device.GetError() ?? "cube-particle upload failed"); + } + + public Result Draw(float cameraX = 0, float cameraY = 0, float jitterX = 0, + float jitterY = 0) + { + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.SetDrawBuffers(target.Framebuffer, 0b111); + device.ClearColor(0, 0.2f, 0.4f, 0.6f, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + device.UseProgram(program); + SetMatrix(device, program, "projectionMatrix", Jittered(jitterX, jitterY)); + SetMatrix(device, program, "modelViewMatrix", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixFar", Identity); + SetMatrix(device, program, "toShadowMapSpaceMatrixNear", Identity); + SetMatrix(device, program, "prevProjectionMatrix", Projection); + SetMatrix(device, program, "prevModelViewMatrix", Identity); + SetFloat3(device, program, "cameraPosDelta", cameraX, cameraY, 0); + SetFloat2(device, program, "taaRenderSize", Size, Size); + SetFloat2(device, program, "taaJitterPx", jitterX, jitterY); + SetFloat3(device, program, "rgbaAmbientIn", 1, 1, 1); + SetInt(device, program, "perceptionEffectId", 1); + SetInt(device, program, "prevPerceptionEffectId", 1); + SetFloat(device, program, "globalWarpIntensity", 0); + SetFloat(device, program, "prevGlobalWarpIntensity", 0); + + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(true); + device.SetDepthMask(true); + device.SetDepthFunc(0x203); // GL_LEQUAL + device.SetCullFace(false); + device.SetBlend(true, EnumBlendMode.Standard); + device.SetBlendFuncSeparate(0, 770, 771, 770, 771); + device.SetBlendEquation(2, 32774); + device.SetBlendFuncSeparate(2, 1, 0, 1, 0); + device.DrawMeshInstanced(mesh, 1); + float[] motion = ReadMotion(device, target.MotionTexture, Size); + byte[] color = device.ReadBackLevel0ForTests(target.ColorTexture); + device.Present(); + return new Result(motion, color); + } + + private static MeshData CubeFace() + { + var mesh = new MeshData(4, 6, withNormals: true, withUv: true, + withRgba: false, withFlags: true) + { + xyz = [-0.5f, -0.5f, 0, 0.5f, -0.5f, 0, + 0.5f, 0.5f, 0, -0.5f, 0.5f, 0], + Uv = [0, 0, 1, 0, 1, 1, 0, 1], + Normals = new int[4], + NormalsCount = 4, + VerticesCount = 4, + Indices = [0, 1, 2, 0, 2, 3], + IndicesCount = 6, + }; + mesh.CustomFloats = new CustomMeshDataPartFloat + { + Instanced = true, + StaticDraw = false, + Values = [0, 0, -1, 1], + Count = 4, + InterleaveSizes = [3, 1], + InterleaveOffsets = [0, 12], + InterleaveStride = 16, + }; + mesh.CustomBytes = new CustomMeshDataPartByte + { + Conversion = DataConversion.NormalizedFloat, + Instanced = true, + StaticDraw = false, + Values = Enumerable.Repeat((byte)255, 12).ToArray(), + Count = 12, + InterleaveSizes = [4, 4, 4], + InterleaveOffsets = [0, 4, 8], + InterleaveStride = 12, + }; + mesh.Flags = new int[4]; + mesh.FlagsInstanced = true; + return mesh; + } + } + + private sealed class MergeScene + { + private readonly VulkanDevice device; + private readonly int program; + private readonly int seedProgram; + private readonly MotionTarget target; + private readonly int quad; + private readonly int accumulation; + private readonly int glow; + private readonly int oitReveal; + private readonly int oitAccumulation; + + public MergeScene(VulkanDevice device) + { + this.device = device; + var variant = ShaderCorpus.Variants().Single(v => v.Name == "taa-no-ssao"); + var stages = ShaderCorpus.BuildProgram("transparentcompose", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + program = LinkFromCorpus(device, stages, "transparentcompose", oit: true); + seedProgram = SeedProgram(device); + target = CreateMotionTarget(device, Size); + quad = FullscreenQuad(device); + accumulation = SolidTexture(device, 0, 0, 0, 0); + glow = SolidTexture(device, 0, 0, 0, 1); + oitReveal = SolidTexture(device, 0, 0, 0, 0); + oitAccumulation = device.CreateTexture2DArray(Size, Size, 3, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba); + } + + public float[] Draw(float revealage, float seedReactive) + { + int reveal = SolidTexture(device, revealage, 0, 0, 1); + device.BeginFrame(); + device.BindFramebuffer(target.Framebuffer); + device.SetDrawBuffers(target.Framebuffer, 0b111); + device.ClearColor(0, 0.2f, 0.4f, 0.6f, 1); + device.ClearColor(1, 0, 0, 0, 1); + device.ClearColor(2, 0, 0, 0, 0); + device.ClearDepth(1); + + device.SetDrawBuffers(target.Framebuffer, 1 << 2); + device.UseProgram(seedProgram); + SetFloat(device, seedProgram, "seedB", seedReactive); + device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(false); + device.SetDepthMask(false); + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.DrawMesh(quad); + + device.SetDrawBuffers(target.Framebuffer, 0b111); + device.SetBlend(true, EnumBlendMode.Standard); + device.SetBlendFuncSeparate(0, 770, 771, 770, 771); + device.SetBlendEquation(2, 32774); + device.SetBlendFuncSeparate(2, 1, 1, 1, 1); + device.UseProgram(program); + Bind("revealage", 10, reveal); + Bind("accumulation", 11, accumulation); + Bind("inGlow", 12, glow); + Bind("OITreveal", 13, oitReveal); + Bind("OITaccumulation", 14, oitAccumulation); + device.DrawMesh(quad); + float[] motion = ReadMotion(device, target.MotionTexture, Size); + device.Present(); + return motion; + } + + private void Bind(string sampler, int unit, int texture) + { + device.SetSamplerUnit(program, sampler, unit); + device.BindTexture(unit, texture); + } + + private static int SeedProgram(VulkanDevice device) + { + const string vertex = "#version 330 core\nlayout(location=0) in vec3 xyz;\n" + + "void main(){gl_Position=vec4(xyz,1);}"; + const string fragment = "#version 330 core\nuniform float seedB;\n" + + "layout(location=2) out vec4 outMotion;\n" + + "void main(){outMotion=vec4(4,-8,seedB,0.5);}"; + return LinkFromCorpus(device, new List + { + new() { Stage = EnumShaderType.VertexShader, Code = vertex, + Filename = "motion-seed.vsh" }, + new() { Stage = EnumShaderType.FragmentShader, Code = fragment, + Filename = "motion-seed.fsh" }, + }, "motion-seed", oit: true); + } + + private static int FullscreenQuad(VulkanDevice device) + { + var quad = new MeshData(4, 6, withNormals: false, withUv: false, + withRgba: false, withFlags: false) + { + xyz = [-1, -1, 0, 1, -1, 0, 1, 1, 0, -1, 1, 0], + VerticesCount = 4, + Indices = [0, 1, 2, 0, 2, 3], + IndicesCount = 6, + }; + int mesh = device.CreateMesh(quad, staticDraw: true); + Assert.True(mesh > 0, device.GetError() ?? "full-screen upload failed"); + return mesh; + } + + private static unsafe int SolidTexture(VulkanDevice device, float r, float g, + float b, float a) + { + byte[] pixels = new byte[Size * Size * 4]; + byte[] value = + [ + (byte)Math.Clamp((int)MathF.Round(r * 255), 0, 255), + (byte)Math.Clamp((int)MathF.Round(g * 255), 0, 255), + (byte)Math.Clamp((int)MathF.Round(b * 255), 0, 255), + (byte)Math.Clamp((int)MathF.Round(a * 255), 0, 255), + ]; + for (int pixel = 0; pixel < pixels.Length; pixel += 4) + value.CopyTo(pixels, pixel); + fixed (byte* source = pixels) + { + int texture = device.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, + (IntPtr)source, false); + device.SetTextureParameter(texture, OptimumGlConstants.TextureMinFilter, 9728); + device.SetTextureParameter(texture, OptimumGlConstants.TextureMagFilter, 9728); + return texture; + } + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeBlitTests.cs b/Optimum.Render.Vulkan.Tests/NativeBlitTests.cs new file mode 100644 index 00000000..31459a13 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeBlitTests.cs @@ -0,0 +1,447 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using OpenTK.Mathematics; +using OpenTK.Windowing.Desktop; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +// The device integration tests' stand-ins for the client's shader types, under names that do +// not read as a device construction to the GPU suite's helper-bypass check. +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The blit to the Default target, drawn twice on one Vulkan device: through the OpenGL body +/// (the route every other vanilla system still takes) and through the native device API +/// (docs/vulkan.md, decision 4). The three branches - TAA debug view, +/// FSR (EASU into the FSR target, RCAS into Default) and the plain blit - have to produce the +/// same pixels, and the native one must not touch the GL state tracker, a texture unit or a +/// draw-buffer mask while its passes are open. +/// +public class NativeBlitTests(ITestOutputHelper output) +{ + private const int WindowSize = 16; + + /// Render resolution below the window: what makes the FSR branch an upsample. + private const int RenderSize = 10; + + private static readonly string[] Programs = { "blit", "fsr-easu", "fsr-rcas", "taa-debug" }; + + /// The Vulkan platform with FSR under the test's control, so no client setting is read. + private sealed class BlitPlatform : VulkanClientPlatform + { + public BlitPlatform() : base(null!) + { + } + + public bool FsrActive; + + public override bool OptimumFsrBlitActive() => FsrActive; + + /// No window is opened here, so both routes take the blit's size from this seam. + public override Size2i OptimumWindowClientSize() => + new(NativeBlitTests.WindowSize, NativeBlitTests.WindowSize); + } + + // ------------------------------------------------------------------ the tests + + /// + /// The plain blit: same pixels, one declared pass, one native draw. + /// + [SkippableFact] + public unsafe void ThePlainBlitMatchesTheOpenGlBodyAsOneNativeDraw() + { + using Session session = Open(); + + byte[] stated = RunFrame(session, native: false, debugView: 0, fsr: false); + + long drawsBefore = session.Seam.NativeDrawsForTests; + long passesBefore = session.Seam.NativePassesForTests; + byte[] nativeRoute = RunFrame(session, native: true, debugView: 0, fsr: false); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + Assert.Equal(stated, nativeRoute); + + GpuTest.AssertClean(session.Seam); + } + + /// Every TAA debug view draws what the OpenGL body draws, with the same mode and render size. + [SkippableTheory] + [InlineData(1)] + [InlineData(2)] + [InlineData(3)] + [InlineData(4)] + public unsafe void ADebugViewMatchesTheOpenGlBody(int mode) + { + using Session session = Open(); + + byte[] stated = RunFrame(session, native: false, debugView: mode, fsr: false); + + long drawsBefore = session.Seam.NativeDrawsForTests; + byte[] nativeRoute = RunFrame(session, native: true, debugView: mode, fsr: false); + + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + Assert.Equal(stated, nativeRoute); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// FSR at a render scale below 1: EASU upsamples Primary colour 0 into the FSR target and + /// RCAS sharpens it into Default - two written targets, two declared passes, and pixels + /// the OpenGL body's two draws agree with. + /// + [SkippableFact] + public unsafe void FsrMatchesTheOpenGlBodyWithOnePassPerWrittenTarget() + { + using Session session = Open(); + + byte[] stated = RunFrame(session, native: false, debugView: 0, fsr: true); + + long drawsBefore = session.Seam.NativeDrawsForTests; + long passesBefore = session.Seam.NativePassesForTests; + byte[] nativeRoute = RunFrame(session, native: true, debugView: 0, fsr: true); + + Assert.Equal(2, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(2, session.Seam.NativeDrawsForTests - drawsBefore); + + int worst = WorstChannelDifference(stated, nativeRoute); + output.WriteLine("FSR worst channel difference: " + worst); + Assert.True(worst <= 1, "FSR differs from the OpenGL body by " + worst + "/255"); + + GpuTest.AssertClean(session.Seam); + } + + /// The OpenGL body on this device is the generic stated route, and the native route is not. + [SkippableFact] + public unsafe void TheOpenGlBodyDrawsThroughTheStatedRouteAndTheNativeRouteDoesNot() + { + using Session session = Open(); + + long nativeDrawsBefore = session.Seam.NativeDrawsForTests; + long statedBefore = session.Platform.StatedDrawsForTests; + RunFrame(session, native: false, debugView: 0, fsr: false); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeDrawsBefore); + Assert.True(session.Platform.StatedDrawsForTests - statedBefore > 0); + + RunFrame(session, native: true, debugView: 0, fsr: false); + + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- driving + + private unsafe byte[] RunFrame(Session session, bool native, int debugView, bool fsr) + { + VulkanDevice seam = session.Seam; + session.Platform.NativeBlitEnabled = native; + session.Platform.FsrActive = fsr; + OptimumConfig.TaaDebugView = debugView; + + session.Platform.BeginFrame(); + + // The motion attachment and the depth the debug views read, written as the frame's + // own clears so both routes see the identical inputs. + seam.BindFramebuffer(session.Primary.FboId); + seam.SetDrawBuffers(session.Primary.FboId, 0b111); + seam.ClearColor(2, 3f, -5f, 0.25f, 0.5f); + seam.ClearDepth(0.5f); + seam.SetDrawBuffers(session.Primary.FboId, 0b011); + + seam.BindDefaultFramebuffer(); + seam.ClearColor(0, 0.125f, 0.75f, 0.375f, 1f); + + // The state ScreenManager.Render leaves the frame in at the blit: blending back on + // after the final composition, no depth test, no culling, viewport on the window. + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetDepthTest(false); + seam.SetCullFace(false); + seam.SetViewport(0, 0, WindowSize, WindowSize); + + session.Platform.BlitPrimaryToDefault(); + + var pixels = new byte[WindowSize * WindowSize * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, WindowSize, WindowSize, (IntPtr)destination); + } + session.Platform.EndFrame(); + return pixels; + } + + private static int WorstChannelDifference(byte[] a, byte[] b) + { + Assert.Equal(a.Length, b.Length); + int worst = 0; + for (int i = 0; i < a.Length; i++) worst = Math.Max(worst, Math.Abs(a[i] - b[i])); + return worst; + } + + // ---------------------------------------------------------------------- session + + private Session Open() + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + + Session? session = Session.TryOpen(output, manifest); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the frame buffers the blit indexes and the shader programs + /// it uses, installed the way the client installs them and put back afterwards. + /// + private sealed class Session : IDisposable + { + public BlitPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + private ShaderProgramBlit? blitBefore; + private ShaderProgram? easuBefore; + private ShaderProgram? rcasBefore; + private ShaderProgram? debugBefore; + private int debugViewBefore; + + public static unsafe Session? TryOpen(ITestOutputHelper output, string manifestDirectory) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-blit-" + Guid.NewGuid().ToString("N")); + var platform = new BlitPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, WindowSize, WindowSize, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + blitBefore = ShaderPrograms.Blit, + easuBefore = ShaderPrograms.FsrEasu, + rcasBefore = ShaderPrograms.FsrRcas, + debugBefore = ShaderPrograms.TaaDebug, + debugViewBefore = OptimumConfig.TaaDebugView, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Primary = CreatePrimary(seam); + FrameBufferRef fsr = CreateFsrTarget(seam); + InstallFrameBuffers(platform, session.Primary, fsr); + + var blit = new ShaderProgramBlit { PassName = "blit" }; + Link(seam, blit, "blit", Array.Empty()); + var easu = new ShaderProgram { PassName = "fsr-easu" }; + Link(seam, easu, "fsr-easu", new[] { "inputTexelSize" }); + var rcas = new ShaderProgram { PassName = "fsr-rcas" }; + Link(seam, rcas, "fsr-rcas", new[] { "inputTexelSize" }); + var debug = new ShaderProgram { PassName = "taa-debug" }; + Link(seam, debug, "taa-debug", new[] { "mode", "renderSize" }); + + ShaderPrograms.Blit = blit; + ShaderPrograms.FsrEasu = easu; + ShaderPrograms.FsrRcas = rcas; + ShaderPrograms.TaaDebug = debug; + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + ShaderPrograms.Blit = blitBefore!; + ShaderPrograms.FsrEasu = easuBefore!; + ShaderPrograms.FsrRcas = rcasBefore!; + ShaderPrograms.TaaDebug = debugBefore!; + OptimumConfig.TaaDebugView = debugViewBefore; + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// Links one vanilla program as ShaderRegistry does and fills the locations its body looks up. + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), new ShaderCorpus.ShaderVariant()); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// Primary at the render resolution: scene, glow and the motion attachment at slot 2, plus depth. + private static unsafe FrameBufferRef CreatePrimary(VulkanDevice seam) + { + var scene = new byte[RenderSize * RenderSize * 4]; + for (int y = 0; y < RenderSize; y++) + { + for (int x = 0; x < RenderSize; x++) + { + int i = (y * RenderSize + x) * 4; + scene[i] = (byte)(20 + x * 23); + scene[i + 1] = (byte)(40 + y * 17); + scene[i + 2] = (byte)(((x ^ y) & 1) * 180 + 30); + scene[i + 3] = 255; + } + } + + int color0; + fixed (byte* pixels = scene) + { + color0 = seam.CreateTexture2D(RenderSize, RenderSize, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + } + + var primary = new FrameBufferRef + { + Width = RenderSize, + Height = RenderSize, + FboId = seam.CreateFramebuffer(RenderSize, RenderSize), + ColorTextureIds = new[] + { + color0, + seam.CreateTexture2D(RenderSize, RenderSize, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + seam.CreateTexture2D(RenderSize, RenderSize, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + DepthTextureId = seam.CreateTexture2D(RenderSize, RenderSize, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false), + }; + for (int slot = 0; slot < primary.ColorTextureIds.Length; slot++) + { + seam.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + primary.ColorTextureIds[slot], 0); + } + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + seam.SetDrawBuffers(primary.FboId, 0b011); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + /// The FSR intermediate at window resolution, as SetupDefaultFrameBuffers builds it. + private static FrameBufferRef CreateFsrTarget(VulkanDevice seam) + { + var target = new FrameBufferRef + { + Width = WindowSize, + Height = WindowSize, + FboId = seam.CreateFramebuffer(WindowSize, WindowSize), + ColorTextureIds = new[] + { + seam.CreateTexture2D(WindowSize, WindowSize, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + }; + seam.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + seam.SetDrawBuffers(target.FboId, 1); + return target; + } + + /// Primary at slot 0 with its motion attachment at 2, the FSR target at 18, and a window to size the blit by. + private static void InstallFrameBuffers(BlitPlatform platform, FrameBufferRef primary, FrameBufferRef fsr) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + list[18] = fsr; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + platform.SetOptimumMotionAttachmentIndex(2); + typeof(ClientPlatformWindows).GetField("optimumTaaTargetsReady", flags)!.SetValue(platform, true); + + // No window: OptimumWindowClientSize is the seam both routes size the blit by, + // and NativeWindow.ClientSize is GLFW-backed, so a stub could not answer it. + } + } + + // ---------------------------------------------------------------- native shaders + + /// The manifest of the four programs the blit uses, built once for the whole class. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + foreach (string program in Programs) + { + NativeShaderBuildResult one = builder.Build(source, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-blit-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeChunkTests.cs b/Optimum.Render.Vulkan.Tests/NativeChunkTests.cs new file mode 100644 index 00000000..b3c70a3b --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeChunkTests.cs @@ -0,0 +1,637 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The terrain, drawn twice on one Vulkan device: through the stated multi-draw the OpenGL +/// body takes (NativeChunksEnabled false) and through the native pass +/// VulkanClientPlatform.NativeChunks.cs records inside a BeginChunkPass / EndChunkPass scope +/// (docs/vulkan.md, decision 5 stage 2). +/// +/// Behavioural identity is the acceptance rule (decision 6). The same chunkopaque program, the +/// same pooled mesh and the same fixed state have to put the same pixels on every attachment of +/// Primary - the scene, the glow and, with a motion window open, the motion attachment bit for +/// bit - across the settings that change the chunk passes: blending on and off, culling on and +/// off, a motion window open and closed, and the motion-only window of the liquid velocity +/// redraw, whose whole point is that it writes the motion attachment and touches nothing else. +/// The shadow cascade, which draws a different program into a different target, gets the same +/// treatment. +/// +public class NativeChunkTests(ITestOutputHelper output) +{ + private const int Size = 32; + + /// The normal-up flags word a solid top face carries (ChunkTerrainRenderTests). + private const int UpNormalFlags = 7 << 18; + + /// The platform with no window: both routes take their size from this seam. + private sealed class ChunkPlatform : VulkanClientPlatform + { + public ChunkPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + // ------------------------------------------------------------------------- the tests + + /// + /// The opaque terrain group: the native route draws what the stated route draws, on every + /// attachment, under each of the blend and cull combinations ChunkRenderer's five Opaque + /// groups run with. + /// + [SkippableTheory] + [InlineData("chunk-opaque", true, true)] + [InlineData("chunk-vegetation", true, false)] + [InlineData("chunk-blendnocull", false, false)] + [InlineData("chunk-decorative", true, true)] + public void ANativeChunkGroupDrawsWhatTheStatedGroupDraws(string pass, bool blend, bool cull) + { + using Session session = Open(motion: false); + + byte[][] stated = session.RunGroup(pass, native: false, blend: blend, cull: cull); + byte[][] native = session.RunGroup(pass, native: true, blend: blend, cull: cull); + + output.WriteLine("scene centre stated " + Centre(stated[0]) + " native " + Centre(native[0])); + Assert.Equal(stated[0], native[0]); + Assert.Equal(stated[1], native[1]); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The native route records the group as one declared pass and one indirect multi-draw - + /// the shape the chunk path has to keep - and the stated route records neither. + /// + [SkippableFact] + public void TheNativeGroupIsOneDeclaredPassAndOneIndirectMultiDraw() + { + using Session session = Open(motion: false); + + long passes = session.Seam.NativePassesForTests; + long indirect = session.Seam.NativeIndirectDrawsForTests; + long draws = session.Seam.NativeDrawsForTests; + session.RunGroup("chunk-opaque", native: false, blend: true, cull: true); + Assert.Equal(0, session.Seam.NativeDrawsForTests - draws); + Assert.Equal(0, session.Seam.NativePassesForTests - passes); + + session.RunGroup("chunk-opaque", native: true, blend: true, cull: true); + Assert.Equal(1, session.Seam.NativePassesForTests - passes); + Assert.Equal(1, session.Seam.NativeIndirectDrawsForTests - indirect); + Assert.Equal(1, session.Seam.NativeDrawsForTests - draws); + GpuTest.AssertClean(session.Seam); + } + + /// + /// With a motion window open, the motion attachment is bit-identical between the two + /// routes: the temporal contract does not change when the group moves to a native pass. + /// + [SkippableFact] + public void TheMotionAttachmentIsIdenticalBetweenTheRoutes() + { + using Session session = Open(motion: true); + + byte[][] stated = session.RunGroup("chunk-opaque", native: false, blend: true, cull: false, motion: true); + byte[][] native = session.RunGroup("chunk-opaque", native: true, blend: true, cull: false, motion: true); + + output.WriteLine("motion centre stated " + Centre(stated[2]) + " native " + Centre(native[2])); + Assert.Equal(stated[0], native[0]); + Assert.Equal(stated[1], native[1]); + Assert.Equal(stated[2], native[2]); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The liquid velocity redraw's window is a colour-write mask, not a draw-buffer toggle: + /// the motion attachment takes the draw and the scene and glow attachments keep exactly the + /// contents the clear left, on both routes. + /// + [SkippableFact] + public void TheMotionOnlyGroupWritesTheMotionAttachmentAndNothingElse() + { + using Session session = Open(motion: true); + + byte[][] stated = session.RunGroup("chunk-liquid-motion", native: false, blend: false, cull: false, + motion: true, motionOnly: true); + byte[][] native = session.RunGroup("chunk-liquid-motion", native: true, blend: false, cull: false, + motion: true, motionOnly: true); + + Assert.Equal(stated[0], native[0]); + Assert.Equal(stated[1], native[1]); + Assert.Equal(stated[2], native[2]); + + // And the mask really is a mask: the shaded slots still hold the clear. + Assert.True(IsClear(native[0], Session.SceneClear), "the motion-only group wrote the scene attachment"); + Assert.True(IsClear(native[1], Session.GlowClear), "the motion-only group wrote the glow attachment"); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The shadow cascade: a different program, a different target, one colour attachment, depth + /// written and no blending. Same acceptance - the two routes agree on the depth the cascade + /// leaves behind, which is the only thing the shadow map is read for. + /// + [SkippableFact] + public void TheShadowCascadeMatchesBetweenTheRoutes() + { + using Session session = Open(motion: false); + + byte[] stated = session.RunShadowGroup(native: false); + byte[] native = session.RunShadowGroup(native: true); + + Assert.Equal(stated, native); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The group's pipelines are built once and kept: a frame of terrain does not rebuild a + /// pipeline per pool, which is what the stated per-draw key resolve used to do. + /// + [SkippableFact] + public void TheGroupBuildsItsPipelinesOnceAndKeepsThem() + { + using Session session = Open(motion: false); + + session.RunGroup("chunk-opaque", native: true, blend: true, cull: true); + int after = session.Seam.NativePipelinesForTests; + session.RunGroup("chunk-opaque", native: true, blend: true, cull: true); + session.RunGroup("chunk-opaque", native: true, blend: true, cull: true); + + Assert.Equal(after, session.Seam.NativePipelinesForTests); + GpuTest.AssertClean(session.Seam); + } + + // ------------------------------------------------------------------------- driving + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + private static bool IsClear(byte[] pixels, byte[] clear) + { + for (int i = 0; i < pixels.Length; i += 4) + { + for (int c = 0; c < 4; c++) + { + if (Math.Abs(pixels[i + c] - clear[c]) > 1) return false; + } + } + return true; + } + + private Session Open(bool motion) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Session? session = Session.TryOpen(output, motion); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the Primary target the Opaque stage binds, a shadow map, the + /// chunk programs and one pooled terrain mesh, installed the way the client installs them + /// and put back afterwards. + /// + private sealed class Session : IDisposable + { + public static readonly byte[] SceneClear = { 32, 64, 128, 255 }; + public static readonly byte[] GlowClear = { 192, 128, 64, 255 }; + + public ChunkPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + public FrameBufferRef Shadow { get; private set; } = null!; + + private ShaderProgram opaque = null!; + private ShaderProgram shadowmap = null!; + private MeshRef pool = null!; + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + private bool motionAttachment; + + /// One pool group: MeshDataPool hands GL's 64-bit byte offsets as int pairs. + private static readonly int[] GroupStarts = { 0, 0 }; + private static readonly int[] GroupSizes = { 6 }; + + public static Session? TryOpen(ITestOutputHelper output, bool motion) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-chunks-" + Guid.NewGuid().ToString("N")); + var platform = new ChunkPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + motionAttachment = motion, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Primary = CreatePrimary(seam, motion ? 3 : 2); + session.Shadow = CreateShadow(seam); + InstallFrameBuffers(platform, session.Primary); + + // The motion attachment is Primary's slot 2 without the SSAO G-buffer, exactly as + // SetupDefaultFrameBuffers publishes it. + platform.SetOptimumMotionAttachmentIndex(motion ? 2 : -1); + + ShaderCorpus.ShaderVariant variant = ShaderCorpus.Variants().First(); + variant.TaaMotion = motion ? 1 : 0; + variant.TaaMotionLocation = 2; + variant.UseSsbo = 0; + session.opaque = session.LinkClientProgram(seam, "chunkopaque", variant); + session.shadowmap = session.LinkClientProgram(seam, "chunkshadowmap", variant); + + session.pool = platform.UploadMesh(BuildBlockFace()); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + // The mesh goes first: VAO's finalizer reaches for ScreenManager.Platform, which is + // about to be the client's again, and a live handle there would crash the test host. + if (pool != null) Platform.DeleteMesh(pool); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// + /// One ChunkRenderer draw group, as the client runs it: Primary bound and cleared, the + /// GL-shaped state the group sets (which the OpenGL body still needs and the native pass + /// ignores), the program's uniforms, then the scope and the pool's multi-draw. + /// + public byte[][] RunGroup(string pass, bool native, bool blend, bool cull, + bool motion = false, bool motionOnly = false) + { + VulkanDevice seam = Seam; + Platform.NativeChunksEnabled = native; + + Platform.BeginFrame(); + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, (1 << Primary.ColorTextureIds.Length) - 1); + Clear(seam); + + Platform.CurrentFrameBuffer = Primary; + seam.SetViewport(0, 0, Size, Size); + seam.UseProgram(opaque.ProgramId); + ShaderProgramBase.CurrentShaderProgram = opaque; + SetProgramUniforms(seam, opaque.ProgramId); + + // What BeginMotionWrite / BeginMotionOnlyWrite do once their guards pass: the window + // flag, the draw-buffer set the stated route needs, and replace blending on the + // motion attachment. The native pass reads the flag and states the rest itself. + SetMotionWriteActive(motion); + if (motion) + { + if (motionOnly) Platform.EnableMotionOnlyDrawBuffers(); + else Platform.EnableMotionDrawBuffers(); + Platform.ApplyOptimumMotionBlendState(); + } + + seam.SetDepthTest(true); + seam.SetDepthMask(true); + seam.SetCullFace(cull); + seam.SetBlend(blend, EnumBlendMode.Standard); + if (blend) Platform.ApplyOptimumMotionBlendState(); + + Platform.BeginChunkPass(pass, blend, depthTest: true, depthWrite: true, cullFace: cull); + try + { + Platform.RenderMesh(pool, GroupStarts, GroupSizes, 1, useSSBOs: false); + } + finally + { + Platform.EndChunkPass(); + } + + if (motion) + { + Platform.RestorePrimaryDrawBuffers(); + SetMotionWriteActive(false); + } + + var attachments = new byte[Primary.ColorTextureIds.Length][]; + for (int slot = 0; slot < attachments.Length; slot++) + { + attachments[slot] = Read(seam, Primary.ColorTextureIds[slot]); + } + Platform.EndFrame(); + return attachments; + } + + /// One shadow cascade group: the shadow map bound, depth written, no blending. + public byte[] RunShadowGroup(bool native) + { + VulkanDevice seam = Seam; + Platform.NativeChunksEnabled = native; + + Platform.BeginFrame(); + seam.BindFramebuffer(Shadow.FboId); + seam.SetDrawBuffers(Shadow.FboId, 1); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearDepth(1f); + + Platform.CurrentFrameBuffer = Shadow; + seam.SetViewport(0, 0, Size, Size); + seam.UseProgram(shadowmap.ProgramId); + ShaderProgramBase.CurrentShaderProgram = shadowmap; + SetProgramUniforms(seam, shadowmap.ProgramId); + + seam.SetDepthMask(true); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.SetDepthTest(true); + seam.SetCullFace(false); + + Platform.BeginChunkPass("chunk-shadow-opaque", blend: false, depthTest: true, depthWrite: true, + cullFace: false); + try + { + Platform.RenderMesh(pool, GroupStarts, GroupSizes, 1, useSSBOs: false); + } + finally + { + Platform.EndChunkPass(); + } + + byte[] pixels = Read(seam, Shadow.ColorTextureIds[0]); + Platform.EndFrame(); + return pixels; + } + + // ----------------------------------------------------------------- the fixtures + + private void Clear(VulkanDevice seam) + { + seam.ClearColor(0, SceneClear[0] / 255f, SceneClear[1] / 255f, SceneClear[2] / 255f, 1f); + seam.ClearColor(1, GlowClear[0] / 255f, GlowClear[1] / 255f, GlowClear[2] / 255f, 1f); + if (Primary.ColorTextureIds.Length > 2) seam.ClearColor(2, 0f, 0f, 0f, 0f); + seam.ClearDepth(1f); + } + + /// The window flag the platform's guards own; a test opens it directly. + private void SetMotionWriteActive(bool active) + { + if (!motionAttachment) return; + typeof(ClientPlatformWindows) + .GetField("optimumMotionWriteActive", BindingFlags.Instance | BindingFlags.NonPublic)! + .SetValue(Platform, active); + } + + /// + /// The uniforms a chunk program needs to draw anything (ChunkTerrainRenderTests: without + /// the view distances every fragment fades out and the pass draws nothing), plus the + /// textures - bound through the platform, because that is the seam the native route + /// takes its handles from and the stated route its units. + /// + private void SetProgramUniforms(VulkanDevice seam, int programId) + { + float[] identity = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + foreach (string name in new[] + { + "projectionMatrix", "modelViewMatrix", "mvpMatrix", + "prevProjectionMatrix", "prevModelViewMatrix", + "toShadowMapSpaceMatrixFar", "toShadowMapSpaceMatrixNear", + }) + { + int location = seam.GetUniformLocation(programId, name); + if (location >= 0) seam.SetUniformMatrix(programId, location, identity); + } + + SetFloat(seam, programId, "viewDistance", 1024f); + SetFloat(seam, programId, "viewDistanceLod0", 1024f); + SetFloat(seam, programId, "alphaTest", 0.001f); + SetFloat(seam, programId, "zNear", 0.1f); + SetFloat(seam, programId, "zFar", 1024f); + SetFloat(seam, programId, "shadowRangeFar", 1024f); + SetFloat(seam, programId, "shadowRangeNear", 64f); + SetFloat(seam, programId, "shadowMapWidthInv", 1f); + SetFloat(seam, programId, "shadowMapHeightInv", 1f); + int ambient = seam.GetUniformLocation(programId, "rgbaAmbientIn"); + if (ambient >= 0) seam.SetUniform(programId, ambient, 1f, 1f, 1f); + int frameSize = seam.GetUniformLocation(programId, "frameSize"); + if (frameSize >= 0) seam.SetUniform(programId, frameSize, (float)Size, (float)Size); + } + + private static void SetFloat(VulkanDevice seam, int programId, string name, float value) + { + int location = seam.GetUniformLocation(programId, name); + if (location >= 0) seam.SetUniform(programId, location, value); + } + + /// + /// Links one vanilla chunk program from the corpus, as ShaderRegistry does, and points + /// every sampler it declares at a small gradient through the platform seam - so a + /// sampling difference between the routes would show as a pixel difference. + /// + private unsafe ShaderProgram LinkClientProgram(VulkanDevice seam, string name, + ShaderCorpus.ShaderVariant variant) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode ?? "", + }; + Assert.True(seam.CompileShader(shader), name + ": " + (seam.GetError() ?? "compile failed")); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + else linked.GeometryShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, name + ": " + (seam.GetError() ?? "link failed")); + + var program = new ShaderProgram { PassName = name, ProgramId = id }; + int unit = 0; + foreach (string sampler in seam.SamplerNamesOf(id)) + { + Platform.BindProgramTexture2D(program, sampler, Gradient(seam, unit), unit); + unit++; + } + return program; + } + + /// One attachment's pixels, read through a framebuffer that holds only it. + private unsafe byte[] Read(VulkanDevice seam, int texture) + { + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + return pixels; + } + + /// Primary as the Opaque stage has it: scene at 0, glow at 1, motion at 2, plus depth. + private static FrameBufferRef CreatePrimary(VulkanDevice seam, int colorCount) + { + var textures = new int[colorCount]; + for (int slot = 0; slot < colorCount; slot++) + { + textures[slot] = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + } + + var primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = textures, + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false), + }; + Attach(seam, primary); + seam.SetDrawBuffers(primary.FboId, (1 << colorCount) - 1); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + /// A shadow cascade's target: one colour attachment and the depth the cascade writes. + private static FrameBufferRef CreateShadow(VulkanDevice seam) + { + var shadow = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false), + }; + Attach(seam, shadow); + seam.SetDrawBuffers(shadow.FboId, 1); + Assert.True(seam.CheckFramebufferComplete(shadow.FboId, out string status), status); + return shadow; + } + + private static void Attach(VulkanDevice seam, FrameBufferRef target) + { + for (int slot = 0; slot < target.ColorTextureIds.Length; slot++) + { + seam.AttachTexture(target.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + target.ColorTextureIds[slot], 0); + } + seam.AttachTexture(target.FboId, EnumFramebufferAttachment.DepthAttachment, target.DepthTextureId, 0); + } + + private static void InstallFrameBuffers(ChunkPlatform platform, FrameBufferRef primary) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + /// A small gradient, so a sampling difference between the routes would show. + private static unsafe int Gradient(VulkanDevice seam, int phase) + { + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + { + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30 + phase * 7); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = 255; + } + } + fixed (byte* first = pixels) + { + return seam.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + /// + /// One tesselated block face in the layout the chunk tesselator emits, covering the + /// middle of the target (ChunkTerrainRenderTests.BuildBlockFace). + /// + private static MeshData BuildBlockFace() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + float[] positions = + { + -0.5f, -0.5f, 0f, + 0.5f, -0.5f, 0f, + 0.5f, 0.5f, 0f, + -0.5f, 0.5f, 0f, + }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags( + positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb, flags: UpNormalFlags); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeEntityDrawTests.cs b/Optimum.Render.Vulkan.Tests/NativeEntityDrawTests.cs new file mode 100644 index 00000000..b0e41e57 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeEntityDrawTests.cs @@ -0,0 +1,620 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// An entity's animated shape, drawn twice on one Vulkan device: through the seam's neutral body +/// (ClientPlatformAbstract.RenderEntityMesh's RenderMesh, the route the OpenGL path takes) and +/// through the native pass VulkanClientPlatform.RenderEntityMesh records +/// (docs/vulkan.md, decision 5 stage 2). +/// +/// Behavioural identity is the acceptance rule (decision 6): the same program, mesh, bone +/// matrices and fixed state have to put the same pixels on Primary's scene and glow attachments +/// AND on the motion attachment, which the temporal contract requires to stay bit-identical. +/// The sweep that matters for this system is the attachment set (with and without the SSAO +/// G-buffer slots, which changes where the motion attachment sits) and the motion window itself +/// (TAA on with the window open, and shut), because the window is a colour write mask on the +/// native route and a draw-buffer toggle on the old one. +/// +public class NativeEntityDrawTests(ITestOutputHelper output) +{ + private const int Size = 16; + + /// The platform with no window: both routes take their size from this seam. + private sealed class EntityPlatform : VulkanClientPlatform + { + public EntityPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + // ------------------------------------------------------------------ the tests + + /// + /// Primary as it is without the SSAO G-buffer (scene, glow, motion): the native entity draw + /// puts the same pixels on all three attachments as the seam's neutral body, and records nothing else. + /// + [SkippableTheory] + [InlineData(false, true)] + [InlineData(false, false)] + [InlineData(true, true)] + [InlineData(true, false)] + public unsafe void TheNativeEntityDrawMatchesTheSeamsNeutralBody(bool gbuffer, bool motionOpen) + { + using Session session = Open(gbuffer); + + byte[][] stated = session.RunFrame(native: false, motionOpen); + + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + byte[][] native = session.RunFrame(native: true, motionOpen); + + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + + for (int slot = 0; slot < stated.Length; slot++) + { + output.WriteLine("slot " + slot + " stated " + Centre(stated[slot]) + + " native " + Centre(native[slot])); + Assert.Equal(stated[slot], native[slot]); + } + // An identity comparison of two blank attachments proves nothing: the shape has to have + // reached the scene slot. + Assert.NotEqual(session.ClearOf(0), Centre(native[0])); + GpuTest.AssertClean(session.Seam); + } + + /// + /// Phase 3b decision 1: a mod renderer stays on the adapter. VSEssentials registers its own + /// entityanimated for the first-person hands, and that program drew the arm wrong through the + /// native route with TAA on; the route therefore takes only the registered vanilla programs. + /// + [SkippableFact] + public void AModRegisteredEntityProgramStaysOnTheNeutralBody() + { + using Session session = Open(gbuffer: false); + session.UnregisterVanillaProgram(); + + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + byte[][] drawn = session.RunFrame(native: true, motionOpen: true); + + Assert.Equal(0, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + Assert.NotEqual(session.ClearOf(0), Centre(drawn[0])); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The motion attachment is the one the temporal contract pins: with the window open the two + /// routes write the same vectors, and with it shut neither route touches it, so whatever was + /// there survives. A native pipeline that forgot the write mask would fail the second half. + /// + [SkippableFact] + public unsafe void TheMotionAttachmentIsIdenticalBetweenTheRoutesAndUntouchedWithTheWindowShut() + { + using Session session = Open(gbuffer: false); + + byte[] statedOpen = session.RunFrame(native: false, motionOpen: true)[Session.MotionSlot]; + byte[] nativeOpen = session.RunFrame(native: true, motionOpen: true)[Session.MotionSlot]; + Assert.Equal(statedOpen, nativeOpen); + + // With the window shut the attachment is out of the draw-buffer set on the old route and + // masked out of the pipeline on the native one, so both leave the frame's clear standing. + // That is rule 9 in its narrowest form: an attachment nothing writes must not pick up + // whatever Vulkan would otherwise leave in it. + byte[] statedShut = session.RunFrame(native: false, motionOpen: false)[Session.MotionSlot]; + byte[] nativeShut = session.RunFrame(native: true, motionOpen: false)[Session.MotionSlot]; + Assert.Equal(statedShut, nativeShut); + Assert.Equal(session.ClearOf(Session.MotionSlot), Centre(nativeShut)); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The seam's neutral body draws through the generic stated route and the native route does not: + /// the switch is real, and "OFF is vanilla" holds for the route the OpenGL path takes. + /// + [SkippableFact] + public unsafe void TheNeutralBodyDrawsThroughTheStatedRouteAndTheNativeRouteDoesNot() + { + using Session session = Open(gbuffer: false); + + long nativeDrawsBefore = session.Seam.NativeDrawsForTests; + session.RunFrame(native: false, motionOpen: true); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeDrawsBefore); + + session.RunFrame(native: true, motionOpen: true); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The pipeline is built once per (program, target, mesh shape, motion window) and kept, and + /// opening or shutting the window is a different pipeline - the write mask is baked into it + /// rather than toggled through a draw-buffer set. + /// + [SkippableFact] + public unsafe void TheEntityPassKeepsItsPipelineAndOneMoreForTheOtherMotionWindow() + { + using Session session = Open(gbuffer: false); + + session.RunFrame(native: true, motionOpen: true); + int afterOpen = session.Seam.NativePipelinesForTests; + session.RunFrame(native: true, motionOpen: true); + Assert.Equal(afterOpen, session.Seam.NativePipelinesForTests); + + session.RunFrame(native: true, motionOpen: false); + Assert.Equal(afterOpen + 1, session.Seam.NativePipelinesForTests); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The animation block is read the same way by both routes. The client uploads the pose once + /// per entity per frame through UBO.Update("Animation", ...); the device snapshots it into + /// the frame's uniform ring per (frame, version) and the draw resolves it when it binds set 2 + /// with its own mesh id. The native draw adds nothing to that - it re-uploads nothing and + /// carries no bone data in its push block - so the two routes have to produce the same image + /// for the same pose, and a pose uploaded before the first draw has to be the one that stands + /// for every later draw of that program. + /// + /// What this does NOT prove, and is recorded rather than asserted: this harness could not make + /// the pose observable in the image. A session posed with a translated bone renders the same + /// pixels as one left at identity, on BOTH routes and with the native manifest variant linked + /// (blocks u:Animation@1, u:AnimationPrev@2). Because both routes agree, it says nothing about + /// this port; it is a question about the animation block's feed on the native shader path, or + /// about this fixture, and it wants a look in the game. + /// + [SkippableFact] + public unsafe void TheAnimationBlockIsReadTheSameWayByBothRoutes() + { + using Session session = Open(gbuffer: false); + session.PoseJoint(shiftX: 0.5f); + + // Warm-up: the native pipeline compiles in the background and its first draws are skipped + // until it is published, exactly as on the stated route. + session.RunFrame(native: true, motionOpen: true); + byte[] posed = session.RunFrame(native: true, motionOpen: true)[0]; + byte[] posedStated = session.RunFrame(native: false, motionOpen: true)[0]; + + Assert.NotEqual(session.ClearOf(0), Centre(posed)); + Assert.Equal(posed, posedStated); + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- driving + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + private Session Open(bool gbuffer) + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + + Session? session = Session.TryOpen(output, manifest, gbuffer); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the Primary target the Opaque stage binds, the entityanimated + /// program with its Animation storage block, and one entity mesh - installed the way the + /// client installs them and put back afterwards. + /// + private sealed class Session : IDisposable + { + /// The motion attachment is the last colour slot of Primary, as SetupDefaultFrameBuffers appends it. + public const int MotionSlot = 2; + + private static readonly float[] Identity = + { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + + public EntityPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + + /// The colour RunFrame clears a slot to, as the readback reports it. + public string ClearOf(int slot) + { + var clear = new[] + { + (byte)Math.Round(0.1f * (slot + 1) * 255f), + (byte)Math.Round(0.2f * 255f), + (byte)Math.Round(0.3f * 255f), + (byte)255, + }; + return clear[0] + "," + clear[1] + "," + clear[2] + "," + clear[3]; + } + + private MeshRef shape = null!; + private ShaderProgram entity = null!; + private ShaderProgramEntityanimated? previousEntityProgram; + + /// What a mod-registered entityanimated looks like to the route: not the registered vanilla program. + public void UnregisterVanillaProgram() => ShaderPrograms.Entityanimated = new ShaderProgramEntityanimated(); + private UBORef animation = null!; + private UBORef animationPrev = null!; + private readonly float[] bones = new float[16 * 4]; + private int atlas; + private bool gbuffer; + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + + private static readonly FieldInfo MotionWriteActive = + typeof(ClientPlatformWindows).GetField("optimumMotionWriteActive", + BindingFlags.Instance | BindingFlags.NonPublic)!; + + public static unsafe Session? TryOpen(ITestOutputHelper output, string manifestDirectory, bool gbuffer) + { + string dataPath = Path.Combine(Path.GetTempPath(), + "optimum-native-entity-" + Guid.NewGuid().ToString("N")); + var platform = new EntityPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + gbuffer = gbuffer, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Primary = CreatePrimary(seam, gbuffer); + InstallFrameBuffers(platform, session.Primary); + // Where SetupDefaultFrameBuffers put the motion attachment: after the shaded set. + platform.SetOptimumMotionAttachmentIndex(session.Primary.ColorTextureIds.Length - 1); + + // The vanilla program type, registered where the client registers it: the native route + // takes vanilla entity programs only, and a mod program under the same pass name stays + // on the neutral body (AModRegisteredEntityProgramStaysOnTheNeutralBody). + var program = new ShaderProgramEntityanimated { PassName = "entityanimated" }; + Link(seam, program, "entityanimated", Variant(gbuffer), new[] + { + "modelMatrix", "viewMatrix", "projectionMatrix", + "rgbaLightIn", "rgbaAmbientIn", "renderColor", "alphaTest", + }); + session.entity = program; + session.previousEntityProgram = ShaderPrograms.Entityanimated; + ShaderPrograms.Entityanimated = program; + + session.atlas = Gradient(seam); + // The two animation blocks ShaderProgramEntityanimated creates for the opaque + // program: the pose and, with TAA on, the previous pose the motion writer reads. + session.animation = platform.CreateUBO(program.ProgramId, 0, "Animation", 16 * 4 * sizeof(float)); + platform.BindUBO((UBO)session.animation); + session.animationPrev = platform.CreateUBO(program.ProgramId, 1, "AnimationPrev", 16 * 4 * sizeof(float)); + platform.BindUBO((UBO)session.animationPrev); + session.PoseJoint(shiftX: 0f); + + session.shape = platform.UploadMesh(BuildShape()); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + ShaderPrograms.Entityanimated = previousEntityProgram!; + if (shape != null) Platform.DeleteMesh(shape); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// + /// The bone matrices, written the way EntityShapeRenderer writes them - through + /// UBO.Update on the "Animation" block, once per entity per frame. Nothing about the + /// native route changes this: the device's storage ring snapshots it per (frame, + /// version), and the draw only has to bind the set with the right mesh id. + /// + public void PoseJoint(float shiftX) + { + for (int joint = 0; joint < 4; joint++) + { + Identity.CopyTo(bones, joint * 16); + bones[joint * 16 + 12] = shiftX; + } + GCHandle pinned = GCHandle.Alloc(bones, GCHandleType.Pinned); + try + { + Platform.UpdateUBO((UBO)animation, pinned.AddrOfPinnedObject(), 0, + bones.Length * sizeof(float), false); + Platform.UpdateUBO((UBO)animationPrev, pinned.AddrOfPinnedObject(), 0, + bones.Length * sizeof(float), false); + } + finally + { + pinned.Free(); + } + } + + /// + /// One frame of the Opaque stage at the point the batched entity loop runs: Primary bound + /// and cleared, the stage's own state set, the program's uniforms and texture bound, the + /// motion window in the state under test, then the seam. + /// + public unsafe byte[][] RunFrame(bool native, bool motionOpen) + { + VulkanDevice seam = Seam; + Platform.NativeEntitiesEnabled = native; + + Platform.BeginFrame(); + int slots = Primary.ColorTextureIds.Length; + uint shadedMask = (1u << (slots - 1)) - 1u; + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, (int)(motionOpen ? (1u << slots) - 1u : shadedMask)); + for (int slot = 0; slot < slots; slot++) + { + seam.ClearColor(slot, 0.1f * (slot + 1), 0.2f, 0.3f, 1f); + } + seam.ClearDepth(1f); + + Platform.CurrentFrameBuffer = Primary; + seam.SetViewport(0, 0, Size, Size); + // What SystemRenderEntities.OnRenderOpaque3D sets before its batched loop. + seam.SetDepthTest(true); + seam.SetDepthMask(true); + seam.SetCullFace(false); + seam.SetBlend(true, EnumBlendMode.Standard); + + seam.UseProgram(entity.ProgramId); + ShaderProgramBase.CurrentShaderProgram = entity; + foreach (string uniform in new[] { "modelMatrix", "viewMatrix", "projectionMatrix" }) + { + seam.SetUniformMatrix(entity.ProgramId, entity.uniformLocations[uniform], Identity); + } + // Lit white with no tint, so the shape actually lands on the attachments: with the + // record left at zero the fragment's alpha is zero and the alpha blend the stage set + // keeps the clear, which would make an identity comparison vacuous. + entity.Uniform("rgbaLightIn", 1f, 1f, 1f, 1f); + entity.Uniform("rgbaAmbientIn", 1f, 1f, 1f); + entity.Uniform("renderColor", 1f, 1f, 1f, 1f); + entity.Uniform("alphaTest", 0.001f); + // The client's own sampler declaration: both routes see the same texture, the + // stated one through the unit and the native one through the declared name. + Platform.BindProgramTexture2D(entity, "entityTex", atlas, 0); + + MotionWriteActive.SetValue(Platform, motionOpen); + // The blend state the motion window forces on its attachment, for the old route. + Platform.ApplyOptimumMotionBlendState(); + + Platform.RenderEntityMesh(shape, "entityTex", atlas); + + MotionWriteActive.SetValue(Platform, false); + + var attachments = new byte[slots][]; + for (int slot = 0; slot < slots; slot++) + { + attachments[slot] = Read(seam, Primary.ColorTextureIds[slot]); + } + Platform.EndFrame(); + return attachments; + } + + /// One attachment's pixels, read through a framebuffer that holds only it. + private unsafe byte[] Read(VulkanDevice seam, int texture) + { + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + seam.BindFramebuffer(Primary.FboId); + return pixels; + } + + // ----------------------------------------------------------------- fixtures + + /// + /// The opaque entity program as ShaderRegistry builds it for this client: not the OIT + /// copy, with the TAA motion writer compiled in at the slot the framebuffer put it, and + /// the G-buffer varyings when the SSAO attachments are there. + /// + private static ShaderCorpus.ShaderVariant Variant(bool gbuffer) => new() + { + UseOit = 0, + TaaMotion = 1, + TaaMotionLocation = gbuffer ? 4 : 2, + SsaoLevel = gbuffer ? 1 : 0, + }; + + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, + ShaderCorpus.ShaderVariant variant, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// + /// Primary as the Opaque stage has it: scene at 0, glow at 1, the SSAO G-buffer's normal + /// and position at 2 and 3 when it is on, and the motion attachment appended after them. + /// + private static FrameBufferRef CreatePrimary(VulkanDevice seam, bool gbuffer) + { + int shaded = gbuffer ? 4 : 2; + var colors = new int[shaded + 1]; + for (int slot = 0; slot < colors.Length; slot++) + { + colors[slot] = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + } + + var primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = colors, + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false), + }; + for (int slot = 0; slot < colors.Length; slot++) + { + seam.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + colors[slot], 0); + } + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + seam.SetDrawBuffers(primary.FboId, (int)((1u << colors.Length) - 1u)); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + private static void InstallFrameBuffers(EntityPlatform platform, FrameBufferRef primary) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + /// A small gradient atlas, so a sampling difference between the routes would show. + private static unsafe int Gradient(VulkanDevice seam) + { + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + { + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = 255; + } + } + fixed (byte* first = pixels) + { + return seam.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + /// + /// The shape as entityanimated sees it: positions, UVs, a per-vertex colour and render + /// flags, reduced to two triangles that cover enough of the target for every attachment + /// to be comparable. damageEffectIn and jointId are left to the layout's constant + /// defaults - jointId 0, the joint PoseJoint moves - which is exactly the GL promise + /// VertexLayoutDescription.WithDefaultsFor keeps for an attribute a mesh does not carry. + /// + private static MeshData BuildShape() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + float[] positions = + { + -0.7f, -0.7f, 0.5f, + 0.7f, -0.7f, 0.5f, + 0.7f, 0.7f, 0.5f, + -0.7f, 0.7f, 0.5f, + }; + float[] uvs = { 0, 0, 1, 0, 1, 1, 0, 1 }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags(positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb, 0); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } + + // ---------------------------------------------------------------- native shaders + + /// The entityanimated program's manifest, built once for the whole class. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + NativeShaderBuildResult one = builder.Build(source, "entityanimated"); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), + "optimum-native-entity-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeGuiTests.cs b/Optimum.Render.Vulkan.Tests/NativeGuiTests.cs new file mode 100644 index 00000000..3748da08 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeGuiTests.cs @@ -0,0 +1,643 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The two GUI and text systems that draw through the native device API, each drawn twice on +/// one Vulkan device: through the seam's neutral body (the OpenGL body's RenderMesh, the route +/// every system that has not moved still takes) and through the native pass +/// VulkanClientPlatform records (docs/vulkan.md, decision 5 stage 2). +/// +/// - the texture-into-texture blit, which bakes every Cairo-drawn GUI and text surface into a +/// texture (ClientMain.RenderTextureIntoFrameBuffer, the texture2texture program); +/// - the aiming reticle's line draws (SystemRenderPlayerAimAcc, the gui program with noTexture +/// set), which are the first native draws with line topology and a caller-chosen line width. +/// +/// Behavioural identity is the acceptance rule (decision 6): the same shader, the same mesh and +/// the same fixed state have to put the same pixels on the target, with blending on and off and +/// at every line width the callers ask for, and the native route must not touch the GL state +/// tracker, a texture unit or a draw-buffer mask while its pass is open. +/// +public class NativeGuiTests(ITestOutputHelper output) +{ + private const int Size = 16; + + /// The platform with no window: both routes take their size from this seam. + private sealed class GuiPlatform : VulkanClientPlatform + { + public GuiPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + // ------------------------------------------------------------------ the tests + + /// + /// The native texture-blit pass draws what the seam's neutral body draws, with blending on + /// (the alphaTest >= 0 case, which is every Cairo bake) and with it off: one declared pass, + /// one native mesh draw, and the same pixels. + /// + [SkippableTheory] + [InlineData(true)] + [InlineData(false)] + public unsafe void TheNativeTextureBlitMatchesTheSeamsNeutralBody(bool blend) + { + using Session session = Open(); + + byte[] stated = session.RunTextureQuad(native: false, blend); + + long passesBefore = session.Seam.NativePassesForTests; + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + byte[] native = session.RunTextureQuad(native: true, blend); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + + output.WriteLine("blit centre stated " + Centre(stated) + " native " + Centre(native)); + Assert.Equal(stated, native); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The native line overlay draws what the seam's neutral body draws, at both widths the + /// aiming reticle asks for - 0.5 for the accuracy rectangle and 1 for the crosshair lines. + /// The clamp to the device's lineWidthRange is on both routes, so a driver whose minimum is + /// 1.0 rasterizes the 0.5 draw the same way through either. + /// + [SkippableTheory] + [InlineData(0.5f)] + [InlineData(1.0f)] + public unsafe void TheNativeLineOverlayMatchesTheSeamsNeutralBody(float lineWidth) + { + using Session session = Open(); + + byte[] stated = session.RunOverlayLines(native: false, lineWidth); + + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + byte[] native = session.RunOverlayLines(native: true, lineWidth); + + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + + output.WriteLine("line row stated " + Row(stated) + " native " + Row(native)); + Assert.Equal(stated, native); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The seams' neutral bodies draw through the generic stated route and the native route does + /// not: the switch is real, and "OFF is vanilla" holds for the route the OpenGL path takes. + /// + [SkippableFact] + public unsafe void TheNeutralBodiesDrawThroughTheStatedRouteAndTheNativeRouteDoesNot() + { + using Session session = Open(); + + long nativeDrawsBefore = session.Seam.NativeDrawsForTests; + long statedBefore = session.Platform.StatedDrawsForTests; + session.RunTextureQuad(native: false, blend: true); + session.RunOverlayLines(native: false, 1.0f); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeDrawsBefore); + Assert.True(session.Platform.StatedDrawsForTests - statedBefore > 0); + + session.RunTextureQuad(native: true, blend: true); + session.RunOverlayLines(native: true, 1.0f); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The two systems are two pipelines and stay two, frame after frame - a Cairo bake that + /// happens hundreds of times in a frame must not build a pipeline per call - and the line + /// width is part of the pipeline's identity, so the reticle's 0.5 and 1.0 draws are two + /// entries rather than one entry drawn twice at whichever width came last. + /// + [SkippableFact] + public unsafe void TheGuiPassesKeepTheirPipelinesAndSeparateTheLineWidths() + { + using Session session = Open(); + + session.RunTextureQuad(native: true, blend: true); + session.RunOverlayLines(native: true, 1.0f); + int afterBoth = session.Seam.NativePipelinesForTests; + + session.RunTextureQuad(native: true, blend: true); + session.RunOverlayLines(native: true, 1.0f); + Assert.Equal(afterBoth, session.Seam.NativePipelinesForTests); + + // A different line width is a different pipeline, not the same one re-emitted. + session.RunOverlayLines(native: true, 0.5f); + Assert.Equal(afterBoth + 1, session.Seam.NativePipelinesForTests); + GpuTest.AssertClean(session.Seam); + } + + /// + /// A Cairo bake is a fresh texture every time. Twenty of them through the native route must + /// cost twenty bindless slot resolutions out of the frame's own arena and still exactly one + /// pipeline: the per-frame descriptor churn the stage brief warns about has to land in the + /// arena, not in a permanent descriptor per texture. + /// + [SkippableFact] + public unsafe void EveryFreshTextureResolvesIntoTheFrameArenaAndBuildsNoNewPipeline() + { + using Session session = Open(); + + session.RunTextureQuad(native: true, blend: true); + int pipelines = session.Seam.NativePipelinesForTests; + + var textures = new List(); + for (int i = 0; i < 20; i++) textures.Add(session.NewGradient(i + 3)); + + long drawsBefore = session.Seam.NativeMeshDrawsForTests; + foreach (int texture in textures) session.RunTextureQuad(native: true, blend: true, texture); + + Assert.Equal(textures.Count, session.Seam.NativeMeshDrawsForTests - drawsBefore); + Assert.Equal(pipelines, session.Seam.NativePipelinesForTests); + GpuTest.AssertClean(session.Seam); + } + + /// + /// An atlas composition samples the texture it writes (BlendedTextureManager copies one atlas + /// region into another region of the same atlas). The native pass takes the same pooled + /// ReadSelf copy the stated route takes, instead of refusing the draw: two native mesh + /// draws, the same pixels, validation clean. + /// + [SkippableFact] + public unsafe void TheNativeTextureBlitReadsItsOwnTargetThroughACopy() + { + using Session session = Open(); + + byte[] stated = session.RunSelfBlit(native: false); + + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + byte[] native = session.RunSelfBlit(native: true); + + Assert.Equal(2, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + + output.WriteLine("self blit row stated " + Row(stated) + " native " + Row(native)); + Assert.Equal(stated, native); + // The right half is the copied left half, not the clear colour. + int left = (Size / 2 * Size + 1) * 4; + int right = (Size / 2 * Size + Size / 2 + 1) * 4; + Assert.Equal(stated[left + 1], stated[right + 1]); + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- driving + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + /// A whole row through the middle, where the overlay's line lands. + private static string Row(byte[] pixels) + { + var text = new System.Text.StringBuilder(); + for (int x = 0; x < Size; x++) + { + int i = (Size / 2 * Size + x) * 4; + text.Append(pixels[i]).Append(':').Append(pixels[i + 3]).Append(' '); + } + return text.ToString(); + } + + private Session Open() + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + + Session? session = Session.TryOpen(output, manifest); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the target the GUI draws into, both programs and both meshes, + /// installed the way the client installs them and put back afterwards. + /// + private sealed class Session : IDisposable + { + private static readonly float[] Identity = + { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + + public GuiPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Target { get; private set; } = null!; + public MeshRef Quad { get; private set; } = null!; + public MeshRef Lines { get; private set; } = null!; + public int SourceTexture { get; private set; } + + private ShaderProgram blit = null!; + private ShaderProgram gui = null!; + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + + public static unsafe Session? TryOpen(ITestOutputHelper output, string manifestDirectory) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-gui-" + Guid.NewGuid().ToString("N")); + var platform = new GuiPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Target = CreateTarget(seam); + InstallFrameBuffers(platform, session.Target); + + var blitProgram = new ShaderProgram { PassName = "texture2texture" }; + Link(seam, blitProgram, "texture2texture", + new[] { "xs", "ys", "width", "height", "texu", "texv", "texw", "texh", "alphaTest" }); + session.blit = blitProgram; + + var guiProgram = new ShaderProgram { PassName = "gui" }; + Link(seam, guiProgram, "gui", + new[] { "projectionMatrix", "modelViewMatrix", "rgbaIn", "noTexture", "applyColor", "alphaTest" }); + session.gui = guiProgram; + + session.SourceTexture = Gradient(seam, 0); + BindSamplerUnits(seam, blitProgram, session.SourceTexture); + BindSamplerUnits(seam, guiProgram, session.SourceTexture); + + session.Quad = platform.UploadMesh(BuildQuad()); + session.Lines = platform.UploadMesh(BuildLines()); + return session; + } + + /// + /// The units the client's program setters bind: what the stated route resolves its + /// samplers through. The native route passes the handles instead. + /// + private static void BindSamplerUnits(VulkanDevice seam, ShaderProgramBase program, int texture) + { + string[] names = seam.SamplerNamesOf(program.ProgramId); + for (int i = 0; i < names.Length; i++) + { + int unit = program.uniformLocations.Count + i; + seam.SetSamplerUnit(program.ProgramId, names[i], unit); + // Only the first sampler ever carries a texture here: texture2texture has one, + // and gui's overlay sampler is unused while noTexture is 1, which is exactly the + // reticle's case - so both routes see nothing bound for it. + seam.BindTexture(unit, i == 0 ? texture : 0); + } + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + // The meshes go first: VAO's finalizer reaches for ScreenManager.Platform, which is + // about to be the client's again, and a live handle there would crash the test host. + if (Quad != null) Platform.DeleteMesh(Quad); + if (Lines != null) Platform.DeleteMesh(Lines); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// A fresh source texture, as a Cairo bake produces one per surface. + public unsafe int NewGradient(int phase) => Gradient(Seam, phase); + + /// + /// One frame at the point RenderTextureIntoFrameBuffer reaches its draw: the + /// destination framebuffer bound and cleared, the depth test off, blending exactly as + /// that method's alphaTest decided, the program's uniforms set, then the seam. + /// + public unsafe byte[] RunTextureQuad(bool native, bool blend, int textureId = 0) + { + VulkanDevice seam = Seam; + Platform.NativeGuiEnabled = native; + int source = textureId != 0 ? textureId : SourceTexture; + + Platform.BeginFrame(); + BeginTarget(seam); + seam.SetBlend(blend, EnumBlendMode.Standard); + + seam.UseProgram(blit.ProgramId); + ShaderProgramBase.CurrentShaderProgram = blit; + // The destination rectangle covers the whole target and the source rectangle the + // whole texture: RenderTextureIntoFrameBuffer's own normalisation, at its limits. + Set(seam, blit, "xs", 0f); + Set(seam, blit, "ys", 0f); + Set(seam, blit, "width", 1f); + Set(seam, blit, "height", 1f); + Set(seam, blit, "texu", 0f); + Set(seam, blit, "texv", 0f); + Set(seam, blit, "texw", 1f); + Set(seam, blit, "texh", 1f); + Set(seam, blit, "alphaTest", blend ? 0.005f : -1f); + if (textureId != 0) seam.BindTexture(blit.uniformLocations.Count, textureId); + + Platform.RenderTextureQuad(Quad, source, blend); + + byte[] pixels = Read(seam); + Platform.EndFrame(); + return pixels; + } + + /// + /// One frame of an atlas composition: the gradient blitted over the whole target, then + /// the target's left half blitted into its right half with the target's own texture as + /// the source, blending off (BlendedTextureManager's base copy). + /// + public unsafe byte[] RunSelfBlit(bool native) + { + VulkanDevice seam = Seam; + Platform.NativeGuiEnabled = native; + + Platform.BeginFrame(); + BeginTarget(seam); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.UseProgram(blit.ProgramId); + ShaderProgramBase.CurrentShaderProgram = blit; + + SetRects(seam, 0f, 1f, 0f, 1f); + seam.BindTexture(blit.uniformLocations.Count, SourceTexture); + Platform.RenderTextureQuad(Quad, SourceTexture, false); + + int own = Target.ColorTextureIds[0]; + SetRects(seam, 0.5f, 0.5f, 0f, 0.5f); + seam.BindTexture(blit.uniformLocations.Count, own); + Platform.RenderTextureQuad(Quad, own, false); + seam.BindTexture(blit.uniformLocations.Count, SourceTexture); + + byte[] pixels = Read(seam); + Platform.EndFrame(); + return pixels; + } + + private void SetRects(VulkanDevice seam, float xs, float width, float texu, float texw) + { + Set(seam, blit, "xs", xs); + Set(seam, blit, "ys", 0f); + Set(seam, blit, "width", width); + Set(seam, blit, "height", 1f); + Set(seam, blit, "texu", texu); + Set(seam, blit, "texv", 0f); + Set(seam, blit, "texw", texw); + Set(seam, blit, "texh", 1f); + Set(seam, blit, "alphaTest", -1f); + } + + /// + /// One frame at the point SystemRenderPlayerAimAcc reaches one of its draws: the gui + /// program current with noTexture set, blending on, the line width it just chose. + /// + public unsafe byte[] RunOverlayLines(bool native, float lineWidth) + { + VulkanDevice seam = Seam; + Platform.NativeGuiEnabled = native; + + Platform.BeginFrame(); + BeginTarget(seam); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetLineWidth(lineWidth); + + seam.UseProgram(gui.ProgramId); + ShaderProgramBase.CurrentShaderProgram = gui; + seam.SetUniformMatrix(gui.ProgramId, gui.uniformLocations["projectionMatrix"], Identity); + seam.SetUniformMatrix(gui.ProgramId, gui.uniformLocations["modelViewMatrix"], Identity); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["rgbaIn"], 1f, 0.5f, 0.25f, 1f); + Set(seam, gui, "noTexture", 1f); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["applyColor"], 0); + Set(seam, gui, "alphaTest", 0f); + + Platform.RenderOverlayLines(Lines, 0, lineWidth, blend: true); + + byte[] pixels = Read(seam); + Platform.EndFrame(); + return pixels; + } + + private static void Set(VulkanDevice seam, ShaderProgramBase program, string name, float value) => + seam.SetUniform(program.ProgramId, program.uniformLocations[name], value); + + /// The target bound, cleared and put into the state both GUI systems draw in. + private void BeginTarget(VulkanDevice seam) + { + seam.BindFramebuffer(Target.FboId); + seam.SetDrawBuffers(Target.FboId, 1); + seam.ClearColor(0, 0.1f, 0.2f, 0.3f, 1f); + + Platform.CurrentFrameBuffer = Target; + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + } + + private unsafe byte[] Read(VulkanDevice seam) + { + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, Target.ColorTextureIds[0], 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + seam.BindFramebuffer(Target.FboId); + return pixels; + } + + // ----------------------------------------------------------------- fixtures + + /// Links one vanilla program as ShaderRegistry does and fills the locations the test sets. + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), new ShaderCorpus.ShaderVariant()); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// + /// The destination a Cairo bake writes into and the default framebuffer the Ortho stage + /// draws into have the same shape here: one colour attachment, no depth. + /// + private static FrameBufferRef CreateTarget(VulkanDevice seam) + { + var target = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + }; + seam.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + seam.SetDrawBuffers(target.FboId, 1); + Assert.True(seam.CheckFramebufferComplete(target.FboId, out string status), status); + return target; + } + + private static void InstallFrameBuffers(GuiPlatform platform, FrameBufferRef target) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = target; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + /// A small gradient, so a sampling difference between the routes would show. + private static unsafe int Gradient(VulkanDevice seam, int phase) + { + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + { + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30 + phase * 7); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = (byte)(96 + ((x + y) & 3) * 40); + } + } + fixed (byte* first = pixels) + { + return seam.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + /// The unit quad ClientMain keeps for its 2D draws: positions and UVs. + private static MeshData BuildQuad() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: false); + float[] positions = + { + -1f, -1f, 0f, + 1f, -1f, 0f, + 1f, 1f, 0f, + -1f, 1f, 0f, + }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags(positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb, 0); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + + /// + /// One line across the middle, in the shape SystemRenderPlayerAimAcc tesselates its + /// reticle in: EnumDrawMode.Lines, positions and a per-vertex colour, no textures. + /// + private static MeshData BuildLines() + { + var mesh = new MeshData(2, 2, withNormals: false, withUv: true, withRgba: true, withFlags: false); + mesh.SetMode(EnumDrawMode.Lines); + mesh.AddVertexWithFlags(-0.8f, 0f, 0f, 0f, 0f, ColorUtil.WhiteArgb, 0); + mesh.AddVertexWithFlags(0.8f, 0f, 0f, 1f, 0f, ColorUtil.WhiteArgb, 0); + mesh.AddIndex(0); + mesh.AddIndex(1); + return mesh; + } + } + + // ---------------------------------------------------------------- native shaders + + /// The two programs' manifest, built once for the whole class. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + foreach (string program in new[] { "texture2texture", "gui" }) + { + NativeShaderBuildResult one = builder.Build(source, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-gui-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativePostChainTests.cs b/Optimum.Render.Vulkan.Tests/NativePostChainTests.cs new file mode 100644 index 00000000..01e194e4 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativePostChainTests.cs @@ -0,0 +1,1560 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The post and TAA chain on the Vulkan platform (docs/vulkan.md, stage 1). +/// Optimum owns the chain's order; its first two passes - the OIT merge and sky motion - draw +/// through the native device API, and the rest run the OpenGL body until their own stage moves +/// them. +/// +/// What is asserted here: +/// 1. the OIT merge draws the same pixels as the OpenGL body, with and without the motion +/// window, into the shaded image, the glow attachment and the motion attachment; +/// 2. sky motion writes the same motion attachment as the OpenGL body and leaves the shaded +/// image alone; +/// 3. each native pass records only its own draws (structural since the GL emulation went); +/// 4. over several frames the chain runs its steps in the declared order and the TAA resolve +/// keeps accumulating - the history parity alternates and the motion attachment the resolve +/// reads was written by the two passes that run before it; +/// 5. the TAA resolve and the TAA sharpen draw the same pixels natively as on the OpenGL body, +/// from a cold history and from a warm one, and the sharpen is skipped on both routes when +/// there is nothing to sharpen; +/// 6. several frames of the native resolve with no readback between them keep accumulating - +/// the result after five frames is not the result after one, and it is the OpenGL body's. +/// +public class NativePostChainTests(ITestOutputHelper output) +{ + private const int Size = 16; + + private static readonly string[] Programs = + { + "transparentcompose", "taa-skymotion", "taa-resolve", "taa-sharpen", "blit", + "findbright", "blur", "godrays", "luma", "final", + }; + + /// + /// A jitter phase per loop step, pinned so two runs of the same length see the same sequence + /// whatever the global frame counter happens to be. The values are Halton-shaped: sub-pixel, + /// never zero, never repeating inside one run. + /// + private static readonly (float X, float Y)[] JitterPhases = + { + (0.25f, -0.375f), (-0.125f, 0.25f), (0.375f, 0.125f), + (-0.375f, -0.25f), (0.125f, 0.375f), (-0.25f, -0.125f), + }; + + /// The Vulkan platform without a window: the size seam answers for one. + private sealed class ChainPlatform : VulkanClientPlatform + { + public ChainPlatform() : base(null!) + { + } + + /// + /// The window the size seam answers with. Render scale below 1 is this size divided by + /// the render targets' size, exactly as the client computes it: client 32 at ssaa 0.5 + /// gives the same 16-pixel targets as client 16 at ssaa 1. + /// + public Size2i ClientSize { get; set; } = new(NativePostChainTests.Size, NativePostChainTests.Size); + + /// + /// The god-rays pass takes its time uniform from here. Pinned so the two routes of a + /// differential run cannot be handed different values by the wall clock. + /// + public override long EllapsedMs => 4242; + + public override Size2i OptimumWindowClientSize() => ClientSize; + + /// + /// No window is opened here, and the base's cases size their viewports from + /// NativeWindow.ClientSize, which is GLFW-backed. Every post target in this fixture is + /// built at exactly the size its case computes - Primary and Luma at the render + /// resolution, the bloom pair at half and at a quarter of it, god rays at half - so + /// binding the target and taking the viewport from it is the same bind and the same + /// viewport, for the OpenGL route and the native one alike. + /// + public override void LoadFrameBuffer(EnumFrameBuffer framebuffer) + { + if (framebuffer == EnumFrameBuffer.Primary) + { + CurrentFrameBuffer = FrameBuffers[0]; + return; + } + switch (framebuffer) + { + case EnumFrameBuffer.BlurHorizontalMedRes: + case EnumFrameBuffer.BlurVerticalMedRes: + case EnumFrameBuffer.BlurHorizontalLowRes: + case EnumFrameBuffer.BlurVerticalLowRes: + case EnumFrameBuffer.GodRays: + CurrentFrameBuffer = FrameBuffers[(int)framebuffer]; + return; + } + base.LoadFrameBuffer(framebuffer); + } + } + + // ------------------------------------------------------------------ the tests + + /// + /// The OIT merge with the motion window open: the shaded image, the glow attachment and the + /// motion attachment all have to come out of the native pass exactly as the OpenGL body + /// leaves them, including the additive (ONE, ONE) blend that only touches the reactive + /// channel. + /// + [SkippableFact] + public void TheOitMergeMatchesTheOpenGlBodyWithTheMotionWindowOpen() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + + Frame stated = RunMerge(session, native: false); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + Frame nativeRoute = RunMerge(session, native: true); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.Equal(stated.Glow, nativeRoute.Glow); + Assert.Equal(stated.Motion, nativeRoute.Motion); + + // The merge really did add into the reactive channel, or the comparison above would + // pass on two routes that both wrote nothing. + Assert.True(Reactive(nativeRoute.Motion, Size / 2, Size / 2) > Reactive(session.MotionSeed, 0, 0), + "the merge did not accumulate coverage into the reactive channel"); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The same merge with TAA off: the motion window never opens, so the pass writes the two + /// world attachments and leaves the motion attachment exactly as it found it. + /// + [SkippableFact] + public void TheOitMergeMatchesTheOpenGlBodyWithoutTheMotionWindow() + { + using Session session = Open(); + session.EnableTaa(jitterActive: false); + + Frame stated = RunMerge(session, native: false); + Frame nativeRoute = RunMerge(session, native: true); + + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.Equal(stated.Glow, nativeRoute.Glow); + Assert.Equal(stated.Motion, nativeRoute.Motion); + Assert.Equal(session.MotionSeed, nativeRoute.Motion); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// Sky motion: the motion attachment the native pass writes is the OpenGL body's, and the + /// shaded image is untouched - the pass writes one colour slot and no depth. + /// + [SkippableFact] + public void SkyMotionMatchesTheOpenGlBody() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + + Frame stated = RunSkyMotion(session, native: false); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + Frame nativeRoute = RunSkyMotion(session, native: true); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(stated.Motion, nativeRoute.Motion); + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.Equal(session.SceneSeed, nativeRoute.Scene); + + // The pass covered the sky, or "the two routes agree" would be vacuous. + Assert.True(Reactive(nativeRoute.Motion, Size / 2, Size / 2) > 0.5f, + "sky motion wrote no reactive value, so the depth test rejected the whole target"); + + GpuTest.AssertClean(session.Seam); + } + + /// The OpenGL body on this device is the generic stated route; the native chain is not. + [SkippableFact] + public void TheOpenGlRouteDrawsThroughTheStatedRouteAndTheNativeChainDoesNot() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + + long nativeDrawsBefore = session.Seam.NativeDrawsForTests; + long statedBefore = session.Platform.StatedDrawsForTests; + RunMerge(session, native: false); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeDrawsBefore); + Assert.True(session.Platform.StatedDrawsForTests - statedBefore > 0); + + RunMerge(session, native: true); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// Several frames through the native chain: every frame runs the steps in the declared + /// order, and TAA keeps accumulating - the resolve runs each frame, the history parity + /// alternates so each frame reads the slot the last one wrote, and the motion attachment it + /// reads carries what the merge and sky motion wrote before it. + /// + /// The final composition is the one declared step not driven here: its body reads + /// NativeWindow.ClientSize directly, which a windowless test cannot answer. Its position in + /// the chain is pinned by the source coverage test (Optimum.Tests, native-post-chain). + /// + [SkippableFact] + public void TheChainKeepsItsOrderAcrossFramesAndTaaKeepsAccumulating() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + + VulkanDevice seam = session.Seam; + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = true; + + var expected = new List(); + foreach (VulkanClientPlatform.NativePostStep step in VulkanClientPlatform.NativePostChainOrder) + { + if (step != VulkanClientPlatform.NativePostStep.FinalComposition) expected.Add(step); + } + + var resolvedTextures = new List(); + const int frames = 4; + for (int frame = 0; frame < frames; frame++) + { + session.AdvanceTemporalFrame(); + var log = new List(); + platform.NativePostStepLog = log; + + platform.BeginFrame(); + session.SeedFrame(); + + platform.CurrentFrameBuffer = session.Primary; + platform.MergeTransparentRenderPass(); + platform.CurrentFrameBuffer = session.Primary; + platform.RenderOptimumSkyMotion(); + platform.RenderPostprocessingEffects(null); + platform.BlitPrimaryToDefault(); + + platform.NativePostStepLog = null; + Assert.Equal(expected, log); + Assert.True(platform.TaaResolvedThisFrame, "the resolve did not run on frame " + frame); + resolvedTextures.Add(ResolvedColorTexture(platform)); + + byte[] motion = session.ReadMotion(); + Assert.True(Reactive(motion, Size / 2, Size / 2) > 0.5f, + "frame " + frame + " reached the resolve with an empty motion attachment, " + + "so the merge and sky motion did not run before it"); + + platform.EndFrame(); + } + + // The resolve alternates its history slots, so every frame reads the one the previous + // frame wrote: that alternation is the accumulation. + int slotA = session.History(0).ColorTextureIds[0]; + int slotB = session.History(1).ColorTextureIds[0]; + for (int frame = 0; frame < frames; frame++) + { + Assert.Equal(frame % 2 == 0 ? slotA : slotB, resolvedTextures[frame]); + } + Assert.True(HistoryValid(platform), "the resolve left the history invalid"); + + GpuTest.AssertClean(seam); + } + + /// + /// The TAA resolve, natively: all three attachments of the history slot it writes - the + /// resolved colour, the resolved glow and the linear depth - have to be the OpenGL body's, + /// from a cold history (the reset frame, which copies the scene through) and from a warm one + /// (the frame that actually blends history in). + /// + /// The temporal invariants are the shader's, and both routes run the same taa-resolve.fsh: what is + /// asserted here is that the native route feeds it the same seven textures and the same nine + /// uniform values, so the 3x3 nearest-depth disocclusion and the luminance anti-flicker + /// weighting see identical inputs and produce identical pixels. + /// + [SkippableTheory] + [InlineData(false)] + [InlineData(true)] + public void TheTaaResolveMatchesTheOpenGlBody(bool warmHistory) + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + session.PatternedScene = true; + + Resolved stated = RunResolve(session, native: false, warmHistory); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + Resolved nativeRoute = RunResolve(session, native: true, warmHistory); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(stated.Color, nativeRoute.Color); + Assert.Equal(stated.Glow, nativeRoute.Glow); + Assert.Equal(stated.Depth, nativeRoute.Depth); + + // The pass wrote a resolved image over the seed, or the comparison above would hold + // for two routes that both wrote nothing. + Assert.NotEqual(session.HistorySeedColor, nativeRoute.Color); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The TAA resolve with TAA off: the lib body returns before any draw, so the native route + /// declares no pass at all and the history is marked invalid on both routes. + /// + [SkippableFact] + public void TheTaaResolveDrawsNothingWithTaaOff() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + OptimumConfig.Taa = false; + + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = true; + long passesBefore = session.Seam.NativePassesForTests; + + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + Assert.False(platform.RenderOptimumTaaResolve()); + platform.EndFrame(); + + Assert.Equal(0, session.Seam.NativePassesForTests - passesBefore); + Assert.False(platform.TaaResolvedThisFrame); + Assert.False(HistoryValid(platform)); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The TAA sharpen, natively: the sharpen target's single attachment has to be the OpenGL + /// body's at every strength the setting can take, and the texture the pass hands on to the + /// rest of the chain has to be the sharpen target either way. + /// + [SkippableTheory] + [InlineData(1f)] + [InlineData(0.35f)] + public void TheTaaSharpenMatchesTheOpenGlBody(float sharpness) + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + session.PatternedScene = true; + OptimumConfig.TaaSharpness = sharpness; + + byte[] stated = RunSharpen(session, native: false, out int statedTexture); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + byte[] nativeRoute = RunSharpen(session, native: true, out int nativeTexture); + + // One native pass for the resolve that has to run first, one for the sharpen. + Assert.Equal(2, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(2, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(session.Sharpen.ColorTextureIds[0], statedTexture); + Assert.Equal(session.Sharpen.ColorTextureIds[0], nativeTexture); + Assert.Equal(stated, nativeRoute); + Assert.NotEqual(session.SharpenSeed, nativeRoute); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The sharpen's conditions live in the lib body, so both routes skip it on exactly the same + /// frames: sharpness at zero hands the resolved texture straight on and draws nothing. + /// + [SkippableFact] + public void TheTaaSharpenIsSkippedOnBothRoutesWhenThereIsNothingToSharpen() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + OptimumConfig.TaaSharpness = 0f; + + ChainPlatform platform = session.Platform; + foreach (bool native in new[] { false, true }) + { + platform.NativePostChainEnabled = native; + long passesBefore = session.Seam.NativePassesForTests; + + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + Assert.True(platform.RenderOptimumTaaResolve()); + int resolved = platform.OptimumPostSceneTexture(); + Assert.Equal(resolved, platform.RenderOptimumTaaSharpen(resolved)); + platform.EndFrame(); + + // The resolve's pass on the native route, and nothing for the sharpen. + Assert.Equal(native ? 1 : 0, session.Seam.NativePassesForTests - passesBefore); + } + + GpuTest.AssertClean(session.Seam); + } + + /// + /// Several frames of the native resolve with nothing read back between them: the history is + /// still being accumulated, not replaced. Five frames do not land where one frame lands - + /// each frame reprojects the previous slot at its own sub-pixel jitter and blends it in - + /// and where they land is the OpenGL body's answer to the identical sequence. + /// + /// A single-frame readback cannot see this: it passed while the R32F-history and + /// masked-clear bugs were live (P2, 2026-09-10), which is why the loop below reads nothing. + /// + [SkippableFact] + public void TheNativeResolveKeepsAccumulatingHistoryAcrossFrames() + { + using Session session = Open(); + session.EnableTaa(jitterActive: true); + session.PatternedScene = true; + + // The same last frame - the same jitter phase, the same scene, the same slot - reached + // two ways: cold, and after four frames of history. Pinning the phase is what makes the + // difference between them history and nothing else. + byte[] lastFrameAlone = RunResolveFrames(session, native: true, frames: 1, startPhase: 4); + byte[] fiveFrames = RunResolveFrames(session, native: true, frames: 5, startPhase: 0); + Assert.NotEqual(lastFrameAlone, fiveFrames); + + byte[] statedFive = RunResolveFrames(session, native: false, frames: 5, startPhase: 0); + Assert.Equal(statedFive, fiveFrames); + + GpuTest.AssertClean(session.Seam); + } + + // ------------------------------------------------- the chain's tail, both routes + + /// + /// The settings that change the chain's tail. Each row is one run of the whole chain on both + /// routes: bloom, god rays, FXAA, vanilla SSAO, TAA, the render scale (client size over target + /// size), the AO debug view and which AO texture the frame produced. + /// + public static TheoryData TailSettings() + { + var data = new TheoryData(); + // name bloom rays fxaa ssao taa ssaa client debug gtao + data.Add("everything-off", false, false, false, false, false, 1f, Size, false, false); + data.Add("bloom", true, false, false, false, false, 1f, Size, false, false); + data.Add("god-rays", false, true, false, false, false, 1f, Size, false, false); + data.Add("fxaa", false, false, true, false, false, 1f, Size, false, false); + data.Add("bloom-rays-fxaa", true, true, true, false, false, 1f, Size, false, false); + data.Add("ssao", true, true, false, true, false, 1f, Size, false, false); + data.Add("ssao-debug-view", false, false, false, true, false, 1f, Size, true, false); + data.Add("ssao-gtao", false, false, false, true, false, 1f, Size, false, true); + data.Add("ssao-gtao-debug", false, false, false, true, false, 1f, Size, true, true); + data.Add("taa", true, true, true, false, true, 1f, Size, false, false); + data.Add("taa-ssao", true, true, true, true, true, 1f, Size, false, false); + data.Add("render-scale-half", true, true, true, true, false, 0.5f, Size * 2, false, false); + return data; + } + + /// + /// Behavioural identity (decision 6) for the four passes this stage made native: the bloom + /// chain, god rays, the Luma step and the final composition. The whole chain runs twice from + /// identically seeded targets - once on the OpenGL body, once natively - and every target the + /// tail writes has to come out the same, bitwise: the find-bright image, the low-resolution + /// bloom result the composition reads, the god-ray target, the Luma target and Primary colour + /// 0. The motion attachment is checked too, because the composition keeps it in its scope + /// while writing colour 0 and must not touch it. + /// + [SkippableTheory] + [MemberData(nameof(TailSettings))] + public void TheChainTailMatchesTheOpenGlBodyAcrossThePostSettings(string name, bool bloom, bool godRays, + bool fxaa, bool ssao, bool taa, float ssaa, int clientSize, bool debugView, bool gtao) + { + using Session session = Open(); + if (taa) session.EnableTaa(jitterActive: true); + else OptimumConfig.Taa = false; + OptimumConfig.AmbientOcclusionDebugView = debugView; + session.ApplyPostSettings(bloom, godRays, fxaa, ssao, ssaa, clientSize); + + int aoTexture = gtao ? session.SsaoBlurTexture : 0; + TailFrame stated = RunTail(session, native: false, aoInScene: gtao, aoTexture: aoTexture); + + long passesBefore = session.Seam.NativePassesForTests; + long copiesBefore = session.Seam.ReadSelfCopiesForTests.Created; + TailFrame nativeRoute = RunTail(session, native: true, aoInScene: gtao, aoTexture: aoTexture); + + // The Luma step and the final composition always draw; bloom adds five passes, god rays + // one, and TAA one more for the resolve (the sharpen declares no pass of its own). No + // native pass reached the generic stated route, and the composition's self-read took no + // feedback copy - the declared attachment subset is what makes it safe. + long expectedPasses = 2 + (bloom ? 5 : 0) + (godRays ? 1 : 0) + (taa ? 1 : 0); + Assert.Equal(expectedPasses, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(copiesBefore, session.Seam.ReadSelfCopiesForTests.Created); + + Assert.Equal(stated.FindBright, nativeRoute.FindBright); + Assert.Equal(stated.BloomLow, nativeRoute.BloomLow); + Assert.Equal(stated.GodRays, nativeRoute.GodRays); + Assert.Equal(stated.Luma, nativeRoute.Luma); + Assert.Equal(stated.Final, nativeRoute.Final); + Assert.Equal(stated.Motion, nativeRoute.Motion); + + // The composition really wrote something, or "the two routes agree" would be vacuous. + Assert.NotEqual(session.SceneSeed, nativeRoute.Final); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The final composition samples Primary colour 1 as the glow on a frame with no TAA resolve, + /// while it writes Primary colour 0. The pass declares the attachment subset, so colour 1 is + /// moved to the shader-read layout for the pass and back afterwards - the frame takes no + /// feedback copy - and the glow attachment itself comes out of the pass unchanged. + /// + [SkippableFact] + public void TheFinalCompositionReadsPrimaryColourOneWithoutAFeedbackCopy() + { + using Session session = Open(); + OptimumConfig.Taa = false; + OptimumConfig.AmbientOcclusionDebugView = false; + session.ApplyPostSettings(bloom: false, godRays: false, fxaa: false, ssao: false, ssaa: 1f, + clientSize: Size); + + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = true; + session.ResetTaaHistory(); + + long copiesBefore = session.Seam.ReadSelfCopiesForTests.Created; + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + Assert.False(platform.TaaResolvedThisFrame); + + platform.RenderFinalComposition(); + + byte[] glow = session.ReadGlow(); + byte[] scene = session.ReadScene(); + platform.EndFrame(); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + Assert.Equal(copiesBefore, session.Seam.ReadSelfCopiesForTests.Created); + Assert.NotEqual(session.SceneSeed, scene); + Assert.Equal(session.GlowSeed, glow); + + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- driving + + private readonly record struct Frame(byte[] Scene, byte[] Glow, byte[] Motion); + + /// The three attachments of the history slot a resolve wrote. + private readonly record struct Resolved(byte[] Color, byte[] Glow, byte[] Depth); + + /// + /// One TAA resolve on the route under test, from an identically seeded frame: the history + /// parity pinned to slot A, both history slots seeded, and the jitter pinned to phase 0 so + /// the two routes see the same sub-pixel offset. + /// + private Resolved RunResolve(Session session, bool native, bool warmHistory) + { + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + session.AdvanceTemporalFrame(0); + SetParity(platform, 0); + SetHistoryValid(platform, warmHistory); + + platform.BeginFrame(); + session.SeedFrame(); + session.SeedHistory(); + platform.CurrentFrameBuffer = session.Primary; + + Assert.True(platform.RenderOptimumTaaResolve(), "the resolve did not run"); + Assert.Equal(1, Parity(platform)); + + var resolved = new Resolved(session.ReadHistoryColor(0), session.ReadHistoryGlow(0), + session.ReadHistoryDepth(0)); + platform.EndFrame(); + return resolved; + } + + /// One resolve and one sharpen on the route under test, from an identically seeded frame. + private byte[] RunSharpen(Session session, bool native, out int handedOn) + { + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + session.AdvanceTemporalFrame(0); + SetParity(platform, 0); + SetHistoryValid(platform, true); + + platform.BeginFrame(); + session.SeedFrame(); + session.SeedHistory(); + session.SeedSharpen(); + platform.CurrentFrameBuffer = session.Primary; + + Assert.True(platform.RenderOptimumTaaResolve(), "the resolve did not run"); + handedOn = platform.RenderOptimumTaaSharpen(platform.OptimumPostSceneTexture()); + + byte[] sharpened = session.ReadSharpen(); + platform.EndFrame(); + return sharpened; + } + + /// + /// resolves on the route under test, with no readback inside the + /// loop - only the frame boundary between them - and the history read once at the end. The + /// run starts cold, so the first frame is the reset frame and every frame after it blends. + /// An odd frame count always ends on slot A, so two runs of different length are comparable. + /// + private byte[] RunResolveFrames(Session session, bool native, int frames, int startPhase) + { + Assert.True(frames % 2 == 1, "an even frame count would end on the other history slot"); + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + SetParity(platform, 0); + SetHistoryValid(platform, false); + + platform.BeginFrame(); + session.SeedHistory(); + platform.EndFrame(); + + for (int frame = 0; frame < frames; frame++) + { + session.AdvanceTemporalFrame(startPhase + frame); + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + Assert.True(platform.RenderOptimumTaaResolve(), "the resolve did not run on frame " + frame); + platform.EndFrame(); + } + + platform.BeginFrame(); + byte[] history = session.ReadHistoryColor(0); + platform.EndFrame(); + return history; + } + + /// Everything the chain's tail writes, on one route. + private readonly record struct TailFrame(byte[] FindBright, byte[] BloomLow, byte[] GodRays, + byte[] Luma, byte[] Final, byte[] Motion); + + /// + /// One whole post chain plus the final composition, on the route under test, from identically + /// seeded targets. The two AO fields are set between the two calls because the chain's first + /// step - the AO step, which is still the OpenGL body on both routes - resets them, exactly as + /// a real frame's GTAO pass would then fill them in. + /// + private TailFrame RunTail(Session session, bool native, bool aoInScene, int aoTexture) + { + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + session.ResetTaaHistory(); + + platform.BeginFrame(); + session.SeedFrame(); + // The history the resolve starts from, so both routes run the chain from identical + // contents; SeedFrame leaves the history alone so frames can accumulate across it. + session.SeedHistory(); + platform.CurrentFrameBuffer = session.Primary; + + platform.RenderPostprocessingEffects(null); + session.ApplyAmbientOcclusionState(aoInScene, aoTexture); + platform.RenderFinalComposition(); + + var frame = new TailFrame( + session.ReadPostTarget(4), + session.ReadPostTarget(8), + session.ReadPostTarget(7), + session.ReadPostTarget(10), + session.ReadScene(), + session.ReadMotion()); + platform.EndFrame(); + return frame; + } + + /// One OIT merge, on the route under test, from an identically seeded frame. + private Frame RunMerge(Session session, bool native) + { + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + + platform.MergeTransparentRenderPass(); + + var frame = new Frame(session.ReadScene(), session.ReadGlow(), session.ReadMotion()); + platform.EndFrame(); + return frame; + } + + /// One sky-motion pass, on the route under test, from an identically seeded frame. + private Frame RunSkyMotion(Session session, bool native) + { + ChainPlatform platform = session.Platform; + platform.NativePostChainEnabled = native; + session.AdvanceTemporalFrame(); + + platform.BeginFrame(); + session.SeedFrame(); + platform.CurrentFrameBuffer = session.Primary; + + bool drawn = platform.RenderOptimumSkyMotion(); + Assert.True(drawn, "the sky motion pass did not run"); + + var frame = new Frame(session.ReadScene(), session.ReadGlow(), session.ReadMotion()); + platform.EndFrame(); + return frame; + } + + /// The reactive channel of a decoded motion pixel (motion.b, in [0, 1]). + private static float Reactive(byte[] decoded, int x, int y) => decoded[(y * Size + x) * 4 + 2] / 255f; + + private static int ResolvedColorTexture(ClientPlatformWindows platform) => + (int)typeof(ClientPlatformWindows) + .GetField("taaResolvedColorTexture", BindingFlags.Instance | BindingFlags.NonPublic)! + .GetValue(platform)!; + + private static int Parity(ClientPlatformWindows platform) => + (int)ParityField.GetValue(platform)!; + + private static void SetParity(ClientPlatformWindows platform, int parity) => + ParityField.SetValue(platform, parity); + + private static void SetHistoryValid(ClientPlatformWindows platform, bool valid) => + HistoryValidField.SetValue(platform, valid); + + private static readonly FieldInfo ParityField = typeof(ClientPlatformWindows) + .GetField("_taaFrameParity", BindingFlags.Instance | BindingFlags.NonPublic)!; + + private static readonly FieldInfo HistoryValidField = typeof(ClientPlatformWindows) + .GetField("_taaHistoryValid", BindingFlags.Instance | BindingFlags.NonPublic)!; + + private static bool HistoryValid(ClientPlatformWindows platform) => + (bool)typeof(ClientPlatformWindows) + .GetField("_taaHistoryValid", BindingFlags.Instance | BindingFlags.NonPublic)! + .GetValue(platform)!; + + // ---------------------------------------------------------------------- session + + private Session Open() + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + Session? session = Session.TryOpen(output, manifest); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the targets the chain indexes, the programs its passes use and + /// the client statics they read, all put back afterwards. + /// + private sealed class Session : IDisposable + { + public ChainPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + public FrameBufferRef Transparent { get; private set; } = null!; + public FrameBufferRef Sharpen { get; private set; } = null!; + + /// + /// True leaves Primary's colour 0 as the pattern it was created with instead of clearing + /// it flat: the TAA resolve's variance clip box collapses on a flat image, so a flat + /// scene would make every frame after the first return the current frame untouched and + /// hide whether history is being blended in at all. + /// + public bool PatternedScene { get; set; } + + /// The history slots' seed and the sharpen target's, as the readbacks decode them. + public byte[] HistorySeedColor { get; private set; } = Array.Empty(); + public byte[] SharpenSeed { get; private set; } = Array.Empty(); + + /// The motion attachment's seed, decoded the way decodes it. + public byte[] MotionSeed { get; private set; } = Array.Empty(); + + /// The shaded image's seed, as read back. + public byte[] SceneSeed { get; private set; } = Array.Empty(); + + /// The glow attachment's seed, decoded the way decodes it. + public byte[] GlowSeed { get; private set; } = Array.Empty(); + + private readonly List buffers = new(); + private int oitReveal; + private int oitAccumulation; + private int scenePattern; + private int decodeProgram; + private int decodeTarget; + private int decodeFramebuffer; + private readonly Dictionary<(int Width, int Height), int> decodeFramebuffers = new(); + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + private ShaderProgramTransparentcompose? composeBefore; + private ShaderProgram? skyMotionBefore; + private ShaderProgram? resolveBefore; + private ShaderProgram? sharpenBefore; + private ShaderProgramBlit? blitBefore; + private ShaderProgramFindbright? findbrightBefore; + private ShaderProgramBlur? blurBefore; + private ShaderProgramGodrays? godraysBefore; + private ShaderProgramLuma? lumaBefore; + private ShaderProgramFinal? finalBefore; + private bool taaBefore; + private float sharpnessBefore; + private bool debugViewBefore; + private object? oitRevealBefore; + private object? oitAccumBefore; + private DefaultShaderUniforms uniforms = new(); + + private const BindingFlags Hidden = BindingFlags.Instance | BindingFlags.NonPublic; + private const BindingFlags HiddenStatic = BindingFlags.Static | BindingFlags.NonPublic; + + public FrameBufferRef History(int parity) => buffers[parity == 0 ? 19 : 20]; + + public static Session? TryOpen(ITestOutputHelper output, string manifestDirectory) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-post-" + Guid.NewGuid().ToString("N")); + var platform = new ChainPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + composeBefore = ShaderPrograms.Transparentcompose, + skyMotionBefore = ShaderPrograms.TaaSkyMotion, + resolveBefore = ShaderPrograms.TaaResolve, + sharpenBefore = ShaderPrograms.TaaSharpen, + blitBefore = ShaderPrograms.Blit, + findbrightBefore = ShaderPrograms.Findbright, + blurBefore = ShaderPrograms.Blur, + godraysBefore = ShaderPrograms.Godrays, + lumaBefore = ShaderPrograms.Luma, + finalBefore = ShaderPrograms.Final, + taaBefore = OptimumConfig.Taa, + sharpnessBefore = OptimumConfig.TaaSharpness, + debugViewBefore = OptimumConfig.AmbientOcclusionDebugView, + }; + ScreenManager.Platform = platform; + ScreenManager.FrameProfiler ??= new FrameProfilerUtil(static (string _) => { }); + platform.ShaderUniforms = session.uniforms; + + session.BuildTargets(); + session.LinkPrograms(); + session.InstallState(); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + ShaderPrograms.Transparentcompose = composeBefore!; + ShaderPrograms.TaaSkyMotion = skyMotionBefore!; + ShaderPrograms.TaaResolve = resolveBefore!; + ShaderPrograms.TaaSharpen = sharpenBefore!; + ShaderPrograms.Blit = blitBefore!; + ShaderPrograms.Findbright = findbrightBefore!; + ShaderPrograms.Blur = blurBefore!; + ShaderPrograms.Godrays = godraysBefore!; + ShaderPrograms.Luma = lumaBefore!; + ShaderPrograms.Final = finalBefore!; + OptimumConfig.Taa = taaBefore; + OptimumConfig.TaaSharpness = sharpnessBefore; + OptimumConfig.AmbientOcclusionDebugView = debugViewBefore; + OptimumTemporal.Frame.JitterActive = false; + typeof(SystemRenderOITLayers).GetField("revealTextureId", HiddenStatic)!.SetValue(null, oitRevealBefore); + typeof(SystemRenderOITLayers).GetField("accumTextureId", HiddenStatic)!.SetValue(null, oitAccumBefore); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + // ------------------------------------------------------------ per-frame state + + /// + /// The state the AfterOIT stage leaves for the merge, and the inputs both routes read: + /// Primary cleared to the seeds, the Transparent target's OIT attachments cleared, the + /// world draw-buffer mask, and the OIT textures on the units the OIT renderer uses. + /// + public void SeedFrame() + { + VulkanDevice seam = Seam; + + seam.BindFramebuffer(Transparent.FboId); + seam.SetDrawBuffers(Transparent.FboId, 0b111001); + seam.ClearColor(0, 0.6f, 0.45f, 0.3f, 1f); + seam.ClearColor(3, 0.30f, 0.10f, 0.05f, 0.5f); + seam.ClearColor(4, 0.10f, 0.25f, 0.05f, 0.35f); + seam.ClearColor(5, 0.05f, 0.10f, 0.30f, 0.2f); + + SeedPostTargets(); + + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, 0b111); + if (!PatternedScene) seam.ClearColor(0, 0.25f, 0.5f, 0.75f, 1f); + seam.ClearColor(1, 0.125f, 0.25f, 0.375f, 1f); + seam.ClearColor(2, 0f, 0f, 0.125f, 0.5f); + seam.ClearDepth(1f); + if (PatternedScene) DrawScenePattern(); + // Primary's default colour set: two attachments without the SSAO G-buffer. + seam.SetDrawBuffers(Primary.FboId, 0b011); + + seam.SetViewport(0, 0, Size, Size); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetDepthTest(false); + seam.SetDepthMask(true); + seam.SetCullFace(false); + + // The units SystemRenderOITLayers points the merge's two OIT samplers at. + ShaderProgramTransparentcompose compose = ShaderPrograms.Transparentcompose; + seam.SetSamplerUnit(compose.ProgramId, "OITreveal", 6); + seam.SetSamplerUnit(compose.ProgramId, "OITaccumulation", 7); + seam.BindTexture(6, oitReveal); + seam.BindTexture(7, oitAccumulation); + } + + /// The post chain's targets, indexed by their EnumFrameBuffer slot. + private static readonly int[] PostTargetIndices = { 2, 3, 4, 7, 8, 9, 10, 14 }; + + /// + /// Every post target seeded to its own constant: the final composition samples the bloom + /// and god-ray targets whether or not their passes ran this frame, so both routes have to + /// start a run from identical contents. The TAA history is NOT seeded here - a frame of + /// the chain must be able to run after another one and find the history the previous + /// frame wrote, which is what the accumulation test measures. Seed it with + /// where a run needs a known starting history. + /// + private void SeedPostTargets() + { + VulkanDevice seam = Seam; + for (int i = 0; i < PostTargetIndices.Length; i++) + { + FrameBufferRef target = buffers[PostTargetIndices[i]]; + if (target == null) continue; + float level = 0.08f + i * 0.09f; + seam.BindFramebuffer(target.FboId); + seam.SetDrawBuffers(target.FboId, 0b1); + seam.ClearColor(0, level, 1f - level, level * 0.5f, 1f); + } + } + + /// + /// The TAA resolve's history slot selection put back to its starting state, so two + /// differential runs of the same chain resolve into the same slot from the same history. + /// + public void ResetTaaHistory() + { + typeof(ClientPlatformWindows).GetField("_taaFrameParity", Hidden)!.SetValue(Platform, 0); + typeof(ClientPlatformWindows).GetField("_taaHistoryValid", Hidden)!.SetValue(Platform, false); + } + + /// + /// The per-frame post switches window_RenderFrame computes, and the render scale: private + /// fields of ClientPlatformWindows, which is where the chain reads them from on both + /// routes (through OptimumRenderBloom and friends on the native one). + /// + public void ApplyPostSettings(bool bloom, bool godRays, bool fxaa, bool ssao, float ssaa, int clientSize) + { + SetField("RenderBloom", bloom); + SetField("RenderGodRays", godRays); + SetField("RenderFXAA", fxaa); + SetField("RenderSSAO", ssao); + SetField("ssaaLevel", ssaa); + Platform.ClientSize = new Size2i(clientSize, clientSize); + } + + /// The two AO fields the final composition reads, as the AO step would have left them. + public void ApplyAmbientOcclusionState(bool inScene, int platformTexture) + { + SetField("optimumSsaoInScene", inScene); + SetField("optimumAmbientOcclusionTexture", platformTexture); + } + + /// The blurred vanilla-SSAO target the final composition binds when AO is on. + public int SsaoBlurTexture => buffers[14].ColorTextureIds[0]; + + private void SetField(string name, object value) => + typeof(ClientPlatformWindows).GetField(name, Hidden)!.SetValue(Platform, value); + + /// + /// The scene pattern into Primary's colour 0, the same every frame: a clear can only + /// write a flat image, and a flat image collapses the resolve's variance clip box. + /// + private void DrawScenePattern() + { + VulkanDevice seam = Seam; + seam.SetDrawBuffers(Primary.FboId, 0b001); + seam.UseProgram(decodeProgram); + seam.SetSamplerUnit(decodeProgram, "source", 15); + seam.BindTexture(15, scenePattern); + SetInt(seam, decodeProgram, "motionMode", 0); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawFullscreenTriangle(); + } + + /// + /// Both history slots to a flat seed, so a resolve starts from known content on either + /// route. Inside a frame: a clear between frames is a no-op on this seam. + /// + public void SeedHistory() + { + VulkanDevice seam = Seam; + for (int parity = 0; parity < 2; parity++) + { + FrameBufferRef slot = History(parity); + seam.BindFramebuffer(slot.FboId); + seam.SetDrawBuffers(slot.FboId, 0b111); + seam.ClearColor(0, 0.9f, 0.1f, 0.4f, 1f); + seam.ClearColor(1, 0.4f, 0.9f, 0.1f, 1f); + seam.ClearColor(2, 12.5f, 0f, 0f, 1f); + } + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, 0b011); + } + + /// The sharpen target to a flat seed, so "the pass wrote something" is checkable. + public void SeedSharpen() + { + VulkanDevice seam = Seam; + seam.BindFramebuffer(Sharpen.FboId); + seam.SetDrawBuffers(Sharpen.FboId, 0b1); + seam.ClearColor(0, 0.05f, 0.95f, 0.55f, 1f); + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, 0b011); + } + + /// TAA on, with the jitter window open or closed, and no sharpen pass. + public void EnableTaa(bool jitterActive) + { + OptimumConfig.Taa = true; + OptimumConfig.TaaSharpness = 0f; + OptimumTemporalFrame frame = OptimumTemporal.Frame; + frame.JitterSequencePx.X = 0.25f; + frame.JitterSequencePx.Y = -0.375f; + frame.JitterActive = jitterActive; + AdvanceTemporalFrame(); + AdvanceTemporalFrame(); + frame.JitterActive = jitterActive; + } + + /// + /// One frame of the temporal contract with the jitter pinned to + /// of , so two runs of the same length are comparable whatever + /// the global frame counter is at. + /// + public void AdvanceTemporalFrame(int phase) + { + AdvanceTemporalFrame(); + OptimumTemporalFrame frame = OptimumTemporal.Frame; + (float x, float y) = JitterPhases[phase % JitterPhases.Length]; + frame.JitterSequencePx.X = x; + frame.JitterSequencePx.Y = y; + // The setter is what copies the sequence into the applied jitter. + frame.JitterActive = true; + } + + /// + /// One frame of the temporal contract: the camera and the projection captured, so the + /// sky-motion and resolve passes have a previous view to reproject through. + /// + public void AdvanceTemporalFrame() + { + OptimumTemporalFrame frame = OptimumTemporal.Frame; + bool jitter = frame.JitterActive; + frame.Advance(16f, Size, Size, 1f, 0.1f, 100f, 70f, uniforms); + double[] projection = Mat4d.Perspective(Mat4d.Create(), 70.0 * Math.PI / 180.0, 1.0, 0.1, 100.0); + double[] view = Mat4d.Identity(Mat4d.Create()); + frame.RecordProjection(EnumTemporalView.World, projection); + frame.CaptureCamera(view, view); + frame.JitterActive = jitter; + } + + // ---------------------------------------------------------------- readback + + public byte[] ReadScene() => ReadAttachmentZero(Primary.FboId, Size, Size); + + public byte[] ReadGlow() => Decode(Primary.ColorTextureIds[1], Size, Size, motion: false); + + public byte[] ReadMotion() => Decode(Primary.ColorTextureIds[2], Size, Size, motion: true); + + public byte[] ReadHistoryColor(int parity) => Decode(History(parity).ColorTextureIds[0], Size, Size, motion: false); + + public byte[] ReadHistoryGlow(int parity) => Decode(History(parity).ColorTextureIds[1], Size, Size, motion: false); + + /// + /// The history's linear depth, through the motion decode: it is an R32F in view-space + /// metres, which the clamping colour decode would flatten to white everywhere. + /// + public byte[] ReadHistoryDepth(int parity) => Decode(History(parity).ColorTextureIds[2], Size, Size, motion: true); + + public byte[] ReadSharpen() => Decode(Sharpen.ColorTextureIds[0], Size, Size, motion: false); + + /// One post target's colour 0, decoded at that target's own size. + public byte[] ReadPostTarget(int index) + { + FrameBufferRef target = buffers[index]; + return Decode(target.ColorTextureIds[0], target.Width, target.Height, motion: false); + } + + private unsafe byte[] ReadAttachmentZero(int framebufferId, int width, int height) + { + var pixels = new byte[width * height * 4]; + fixed (byte* destination = pixels) + { + Seam.BindFramebuffer(framebufferId); + Seam.ReadDefaultFramebuffer(0, 0, width, height, (IntPtr)destination); + } + return pixels; + } + + /// + /// Any attachment through an RGBA8 copy of its own size, because the seam's readback is + /// four bytes per pixel from attachment 0. The motion mode encodes the vector into the two + /// low channels so a difference in it cannot hide behind a clamp. + /// + private byte[] Decode(int textureId, int width, int height, bool motion) + { + VulkanDevice seam = Seam; + int framebuffer = DecodeFramebuffer(width, height); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.UseProgram(decodeProgram); + seam.SetSamplerUnit(decodeProgram, "source", 15); + seam.BindTexture(15, textureId); + SetInt(seam, decodeProgram, "motionMode", motion ? 1 : 0); + seam.SetViewport(0, 0, width, height); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawFullscreenTriangle(); + return ReadAttachmentZero(framebuffer, width, height); + } + + /// The RGBA8 decode target of one size, made once. + private int DecodeFramebuffer(int width, int height) + { + if (width == Size && height == Size) return decodeFramebuffer; + if (decodeFramebuffers.TryGetValue((width, height), out int existing)) return existing; + + int texture = Seam.CreateTexture2D(width, height, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int framebuffer = Seam.CreateFramebuffer(width, height); + Seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + Seam.SetDrawBuffers(framebuffer, 0b1); + decodeFramebuffers[(width, height)] = framebuffer; + return framebuffer; + } + + private static void SetInt(VulkanDevice seam, int program, string name, int value) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, value); + } + + // ------------------------------------------------------------------- fixture + + private void BuildTargets() + { + VulkanDevice seam = Seam; + + Primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + Texture(EnumTextureInternalFormat.Rgba8), + Texture(EnumTextureInternalFormat.Rgba8), + Texture(EnumTextureInternalFormat.Rgba16f), + }, + DepthTextureId = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.DepthComponent32, + EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false), + }; + for (int slot = 0; slot < 3; slot++) + { + seam.AttachTexture(Primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + Primary.ColorTextureIds[slot], 0); + } + seam.AttachTexture(Primary.FboId, EnumFramebufferAttachment.DepthAttachment, Primary.DepthTextureId, 0); + Assert.True(seam.CheckFramebufferComplete(Primary.FboId, out string primaryStatus), primaryStatus); + + // The Transparent target as the client leaves it once the OIT renderer has replaced + // attachment 0 with its reveal target and attached the accumulation array's three + // layers at 3, 4 and 5. ColorTextureIds keeps the vanilla ids, which is what the + // merge binds as accumulation, revealage and in-glow. + oitReveal = Texture(EnumTextureInternalFormat.Rgba8); + oitAccumulation = seam.CreateTexture2DArray(Size, Size, 3, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba); + Transparent = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + Seeded(0.7f), + Seeded(0.35f), + Seeded(0.2f), + }, + }; + seam.AttachTexture(Transparent.FboId, EnumFramebufferAttachment.ColorAttachment0, oitReveal, 0); + seam.AttachTexture(Transparent.FboId, EnumFramebufferAttachment.ColorAttachment1, + Transparent.ColorTextureIds[1], 0); + seam.AttachTexture(Transparent.FboId, EnumFramebufferAttachment.ColorAttachment2, + Transparent.ColorTextureIds[2], 0); + seam.AttachTexture(Transparent.FboId, EnumFramebufferAttachment.ColorAttachment3, oitAccumulation, 0); + seam.AttachTexture(Transparent.FboId, EnumFramebufferAttachment.ColorAttachment4, oitAccumulation, 1); + seam.AttachTexture(Transparent.FboId, (EnumFramebufferAttachment)36069, oitAccumulation, 2); + Assert.True(seam.CheckFramebufferComplete(Transparent.FboId, out string status), status); + + for (int i = 0; i <= 24; i++) buffers.Add(null!); + buffers[0] = Primary; + buffers[1] = Transparent; + // The post chain's targets at the sizes and formats SetupDefaultFrameBuffers builds + // them at: the bloom ping-pongs at half and quarter resolution, find-bright, god rays + // and Luma at full, and the blurred vanilla-SSAO target the final composition binds. + buffers[2] = SingleTarget(Size / 2, Size / 2, EnumTextureInternalFormat.Rgba8); + buffers[3] = SingleTarget(Size / 2, Size / 2, EnumTextureInternalFormat.Rgba8); + buffers[4] = SingleTarget(Size, Size, EnumTextureInternalFormat.Rgba16f); + buffers[7] = SingleTarget(Size / 2, Size / 2, EnumTextureInternalFormat.Rgba16f); + buffers[8] = SingleTarget(Size / 4, Size / 4, EnumTextureInternalFormat.Rgba8); + buffers[9] = SingleTarget(Size / 4, Size / 4, EnumTextureInternalFormat.Rgba8); + buffers[10] = SingleTarget(Size, Size, EnumTextureInternalFormat.Rgba16f); + buffers[14] = SingleTarget(Size, Size, EnumTextureInternalFormat.Rgba8); + buffers[19] = HistoryTarget(); + buffers[20] = HistoryTarget(); + Sharpen = SingleTarget(Size, Size, EnumTextureInternalFormat.Rgba16f); + buffers[21] = Sharpen; + + scenePattern = Seeded(0.45f); + decodeTarget = Texture(EnumTextureInternalFormat.Rgba8); + decodeFramebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(decodeFramebuffer, EnumFramebufferAttachment.ColorAttachment0, decodeTarget, 0); + seam.SetDrawBuffers(decodeFramebuffer, 0b1); + } + + private int Texture(EnumTextureInternalFormat format) => + Seam.CreateTexture2D(Size, Size, format, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + + /// A texture with a per-pixel pattern around , so a difference shows. + private unsafe int Seeded(float level) + { + var pixels = new byte[Size * Size * 4]; + for (int y = 0; y < Size; y++) + { + for (int x = 0; x < Size; x++) + { + int i = (y * Size + x) * 4; + pixels[i] = (byte)Math.Clamp(level * 255f + x * 3, 0, 255); + pixels[i + 1] = (byte)Math.Clamp(level * 255f + y * 5, 0, 255); + pixels[i + 2] = (byte)Math.Clamp(level * 255f + ((x ^ y) & 7) * 9, 0, 255); + pixels[i + 3] = (byte)Math.Clamp(level * 255f, 0, 255); + } + } + fixed (byte* data = pixels) + { + return Seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)data, false); + } + } + + private FrameBufferRef SingleTarget(int width, int height, EnumTextureInternalFormat format) + { + var target = new FrameBufferRef + { + Width = width, + Height = height, + FboId = Seam.CreateFramebuffer(width, height), + ColorTextureIds = new[] + { + Seam.CreateTexture2D(width, height, format, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + }; + Seam.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + Seam.SetDrawBuffers(target.FboId, 0b1); + return target; + } + + /// A TAA history slot: colour, aux and the linear-depth R32F, as the platform builds them. + private FrameBufferRef HistoryTarget() + { + var target = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = Seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + Texture(EnumTextureInternalFormat.Rgba16f), + Texture(EnumTextureInternalFormat.Rgba8), + Seam.CreateTexture2DRaw(Size, Size, 0x822E, IntPtr.Zero, 0), + }, + }; + for (int slot = 0; slot < 3; slot++) + { + Seam.AttachTexture(target.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + target.ColorTextureIds[slot], 0); + } + Seam.SetDrawBuffers(target.FboId, 0b111); + return target; + } + + private void LinkPrograms() + { + VulkanDevice seam = Seam; + ShaderCorpus.ShaderVariant variant = TaaVariant(); + + var compose = new ShaderProgramTransparentcompose { PassName = "transparentcompose" }; + Link(seam, compose, "transparentcompose", variant, Array.Empty()); + var skyMotion = new ShaderProgram { PassName = "taa-skymotion" }; + Link(seam, skyMotion, "taa-skymotion", variant, new[] + { + "taaRenderSize", "taaJitterPx", "taaInvViewProjJittered", "taaPrevViewProj", "taaCloudReactive", + }); + var sharpen = new ShaderProgram { PassName = "taa-sharpen" }; + Link(seam, sharpen, "taa-sharpen", variant, new[] { "inputTexelSize", "sharpness" }); + var resolve = new ShaderProgram { PassName = "taa-resolve" }; + Link(seam, resolve, "taa-resolve", variant, new[] + { + "renderSize", "jitterPx", "invViewProjJittered", "prevViewProj", "viewMatrix", + "cameraDelta", "resetHistory", "blendAlpha", "varianceGamma", + }); + var blit = new ShaderProgramBlit { PassName = "blit" }; + Link(seam, blit, "blit", variant, Array.Empty()); + + // The chain's tail: find-bright, the blur ladder, god rays, the FXAA luma prepass and + // the final composition, with every uniform the OpenGL body sets on them registered + // so the old route can run too. + var findbright = new ShaderProgramFindbright { PassName = "findbright" }; + Link(seam, findbright, "findbright", variant, new[] { "ambientBloomLevel", "extraBloom" }); + var blur = new ShaderProgramBlur { PassName = "blur" }; + Link(seam, blur, "blur", variant, new[] { "frameSize", "isVertical" }); + var godrays = new ShaderProgramGodrays { PassName = "godrays" }; + Link(seam, godrays, "godrays", variant, new[] + { + "invFrameSizeIn", "maxGodRaySamples", "sunPosScreenIn", "sunPos3dIn", + "playerViewVector", "dusk", "iGlobalTimeIn", + }); + var luma = new ShaderProgramLuma { PassName = "luma" }; + Link(seam, luma, "luma", variant, Array.Empty()); + var final = new ShaderProgramFinal { PassName = "final" }; + Link(seam, final, "final", variant, new[] + { + "ambientBloomLevel", "optimumSsaoInScene", "optimumAoDebug", "invFrameSizeIn", + "gammaLevel", "extraGamma", "contrastLevel", "brightnessLevel", "sepiaLevel", + "windWaveCounter", "glitchEffectStrength", "sunPosScreenIn", "sunPos3dIn", + "playerViewVector", "damageVignetting", "damageVignettingSide", "frostVignetting", + }); + + ShaderPrograms.Transparentcompose = compose; + ShaderPrograms.TaaSkyMotion = skyMotion; + ShaderPrograms.TaaResolve = resolve; + ShaderPrograms.TaaSharpen = sharpen; + ShaderPrograms.Blit = blit; + ShaderPrograms.Findbright = findbright; + ShaderPrograms.Blur = blur; + ShaderPrograms.Godrays = godrays; + ShaderPrograms.Luma = luma; + ShaderPrograms.Final = final; + + decodeProgram = LinkDecode(seam); + } + + /// TAA on without the SSAO G-buffer: the motion attachment at colour 2. + internal static ShaderCorpus.ShaderVariant TaaVariant() + { + foreach (ShaderCorpus.ShaderVariant candidate in ShaderCorpus.Variants()) + { + if (candidate.Name == "taa-no-ssao") return candidate; + } + throw new InvalidOperationException("the corpus has no taa-no-ssao variant"); + } + + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, + ShaderCorpus.ShaderVariant variant, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// The readback helper's own program: any attachment into RGBA8. + private static int LinkDecode(VulkanDevice seam) + { + const string vertex = @"#version 330 core +out vec2 uv; +void main(void) +{ + vec2 position = vec2((gl_VertexID << 1) & 2, gl_VertexID & 2); + uv = position; + gl_Position = vec4(position * 2.0 - 1.0, 0.0, 1.0); +} +"; + const string fragment = @"#version 330 core +uniform sampler2D source; +uniform int motionMode; +layout(location = 0) out vec4 outColor; +void main(void) +{ + vec4 texel = texelFetch(source, ivec2(gl_FragCoord.xy), 0); + if (motionMode != 0) { + outColor = vec4( + clamp(texel.r / 32.0 * 0.5 + 0.5, 0.0, 1.0), + clamp(texel.g / 32.0 * 0.5 + 0.5, 0.0, 1.0), + clamp(texel.b, 0.0, 1.0), + clamp(texel.a, 0.0, 1.0)); + return; + } + outColor = clamp(texel, 0.0, 1.0); +} +"; + var linked = new LinkedProgram { PassName = "native-post-decode" }; + var vertexShader = new LinkedShader { Type = EnumShaderType.VertexShader, Code = vertex, PrefixCode = "" }; + var fragmentShader = new LinkedShader { Type = EnumShaderType.FragmentShader, Code = fragment, PrefixCode = "" }; + Assert.True(seam.CompileShader(vertexShader), seam.GetError() ?? "decode vertex shader"); + Assert.True(seam.CompileShader(fragmentShader), seam.GetError() ?? "decode fragment shader"); + linked.VertexShader = vertexShader; + linked.FragmentShader = fragmentShader; + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "decode link failed"); + return id; + } + + /// The platform's frame buffers, motion attachment and TAA readiness, as the setup leaves them. + private void InstallState() + { + typeof(ClientPlatformWindows).GetField("frameBuffers", Hidden)!.SetValue(Platform, buffers); + typeof(ClientPlatformWindows).GetField("ssaaLevel", Hidden)!.SetValue(Platform, 1f); + Platform.SetOptimumMotionAttachmentIndex(2); + typeof(ClientPlatformWindows).GetField("optimumTaaTargetsReady", Hidden)!.SetValue(Platform, true); + + FieldInfo reveal = typeof(SystemRenderOITLayers).GetField("revealTextureId", HiddenStatic)!; + FieldInfo accumulation = typeof(SystemRenderOITLayers).GetField("accumTextureId", HiddenStatic)!; + oitRevealBefore = reveal.GetValue(null); + oitAccumBefore = accumulation.GetValue(null); + reveal.SetValue(null, oitReveal); + accumulation.SetValue(null, oitAccumulation); + + // The seeds the comparisons quote, read once through the same decode the tests use. + Platform.BeginFrame(); + SeedFrame(); + SeedHistory(); + SeedSharpen(); + SceneSeed = ReadScene(); + GlowSeed = ReadGlow(); + MotionSeed = ReadMotion(); + HistorySeedColor = ReadHistoryColor(0); + SharpenSeed = ReadSharpen(); + Platform.EndFrame(); + } + } + + // ---------------------------------------------------------------- native shaders + + /// The manifest of the programs the chain's two native passes and the resolve use. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + foreach (string program in Programs) + { + NativeShaderBuildResult one = builder.Build(source, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-post-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeShaderParityTests.cs b/Optimum.Render.Vulkan.Tests/NativeShaderParityTests.cs new file mode 100644 index 00000000..54e53570 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeShaderParityTests.cs @@ -0,0 +1,555 @@ +using System; +using System.Collections.Generic; +using System.Diagnostics; +using System.Globalization; +using System.IO; +using System.Linq; +using System.Text.RegularExpressions; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Xunit; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// Static parity between every native program in sources/shaders-vk and the GLSL 330 program it +/// replaces (docs/vulkan.md). Data-driven over the tree: a family stage adds +/// shaders, never test code. +/// +/// The native side is the manifest the offline compiler's library entry point +/// () produces for the tree. The GLSL 330 side is the program +/// builds from the effective sources (sources/shaders override, else +/// the vanilla asset) with the prefix defines of the matching variant. No GPU is needed. +/// +public sealed class NativeShaderParityTests +{ + private static string SourceDirectory => Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + + private static readonly Lazy<(NativeShaderBuildResult? Result, string Reason)> Built = new(() => + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return (null, reason); + using (compiler) + { + return (new NativeShaderBuilder(compiler!).Build(SourceDirectory), ""); + } + }); + + public static IEnumerable Programs() => + NativeShaderBuilder.DiscoverPrograms(SourceDirectory, new List()).Select(name => new object[] { name }); + + private static NativeShaderBuildResult RequireBuild() + { + (NativeShaderBuildResult? result, string reason) = Built.Value; + Skip.If(result == null, reason); + return result!; + } + + private static List ErrorsOf(NativeShaderBuildResult result, string program) => + result.Errors.Where(e => e.StartsWith(program + " ", StringComparison.Ordinal) || e.StartsWith(program + ":", StringComparison.Ordinal)).ToList(); + + // ------------------------------------------------------------------ the GLSL 330 oracle + + /// + /// ShaderProgram.collectUniformNames (build/VintagestoryLib/Vintagestory.Client.NoObf/ShaderProgram.cs:57-68), + /// character for character: the same pattern and options, run over each stage's include-expanded + /// Code (never the prefix, never preprocessed) in the order Compile calls it (vertex, + /// fragment, geometry). It therefore sees names inside inactive #if blocks and comments, and a + /// type outside its list is invisible to it. + /// + private static readonly Regex CollectUniformNames = new( + "(\\s|\\r\\n)uniform\\s*(?float|int|ivec2|ivec3|ivec4|vec2|vec3|vec4|sampler2DShadow|sampler2D|samplerCube|mat3|mat4x3|mat4)\\s*(\\[[\\d\\w]+\\])?\\s*(?[\\d\\w]+)", + RegexOptions.IgnoreCase | RegexOptions.ExplicitCapture); + + internal sealed class Oracle + { + /// The name set, each with every type the pattern captured for it. + public readonly SortedDictionary> Names = new(StringComparer.Ordinal); + /// textureLocations: a repeated sampler name is reassigned the current count, as the client does. + public readonly Dictionary TextureLocations = new(StringComparer.Ordinal); + /// + /// The pattern has no sampler2DArray, so uniform sampler2DArray OITaccumulation matches as type + /// sampler2D named Array. The client registers that name and unit; the program's real sampler is the + /// declared one. Key: the name the client sees; value: the declared sampler2DArray name. The runtime answers + /// the client's name with the declared sampler's slot. + /// + public readonly Dictionary ArraySamplerAliases = new(StringComparer.Ordinal); + } + + private static readonly Regex DeclaredSampler2DArray = new(@"\G(\s|\r\n)uniform\s*sampler2DArray\s+(?[\d\w]+)", + RegexOptions.IgnoreCase | RegexOptions.ExplicitCapture); + + internal static Oracle CollectOracle(IEnumerable stages) + { + var oracle = new Oracle(); + foreach (ShaderStageSource stage in stages.OrderBy(s => s.Stage switch + { + EnumShaderType.VertexShader => 0, + EnumShaderType.FragmentShader => 1, + _ => 2, + })) + { + foreach (Match item in CollectUniformNames.Matches(stage.Code)) + { + string value = item.Groups["var"].Value; + string type = item.Groups["type"].ToString(); + Match array = DeclaredSampler2DArray.Match(stage.Code, item.Index); + if (value == "Array" && array.Success) + { + oracle.ArraySamplerAliases["Array"] = array.Groups["var"].Value; + value = array.Groups["var"].Value; + type = "sampler2DArray"; + } + if (!oracle.Names.TryGetValue(value, out SortedSet? types)) + { + oracle.Names[value] = types = new SortedSet(StringComparer.Ordinal); + } + types.Add(type); + if (type.Contains("sampler")) + { + oracle.TextureLocations[value] = oracle.TextureLocations.Count; + } + } + } + return oracle; + } + + // ------------------------------------------------------------------ variants + + /// + /// The defines every value of every axis is checked under. Axes a native program branches on override + /// the base; everything else keeps the base's value, so a define that changes a GLSL 330 declaration + /// without being an axis of the native program makes the bases disagree and fails. + /// + private static readonly string[] BaseVariants = { "everything-off", "everything-on", "taa-with-ssao" }; + + internal static Dictionary ParseKey(string key) + { + var values = new Dictionary(StringComparer.Ordinal); + if (key.Length == 0) return values; + foreach (string pair in key.Split(',')) + { + string[] parts = pair.Split('='); + values[parts[0]] = int.Parse(parts[1], CultureInfo.InvariantCulture); + } + return values; + } + + /// + /// Maps a native variant back to ShaderRegistry's prefix defines (registerDefaultShaderCodePrefixes, + /// ShaderRegistry.cs:462-540). GBUFFER is SSAOLEVEL > 0; TAAMOTIONLOCATION follows it (4 with the + /// G-buffer, 2 without); the per-registration defines are stamped as the program's own prefix, which + /// the GLSL 330 sources test with defined(). + /// + internal static ShaderCorpus.ShaderVariant VariantFor(string baseName, IReadOnlyDictionary axes) + { + ShaderCorpus.ShaderVariant variant = ShaderCorpus.Variants().Single(v => v.Name == baseName); + var extra = new List(); + foreach ((string axis, int value) in axes.OrderBy(a => a.Key, StringComparer.Ordinal)) + { + switch (axis) + { + case "TAAMOTION": variant.TaaMotion = value; break; + case "GBUFFER": variant.SsaoLevel = value == 0 ? 0 : Math.Max(1, variant.SsaoLevel); break; + case "USEOIT": variant.UseOit = value; break; + case "USESSBO": variant.UseSsbo = value; break; + case "GREEDYMESH": variant.GreedyMesh = value; break; + case "ALLOWDEPTHOFFSET" or "GLOWSUB" or "VEC3SCALE": + if (value != 0) extra.Add("#define " + axis + " 1"); + break; + default: throw new InvalidOperationException("no GLSL 330 mapping for axis " + axis); + } + } + variant.TaaMotionLocation = variant.SsaoLevel > 0 ? 4 : 2; + variant.ExtraPrefix = string.Join("\r\n", extra); + variant.Name = baseName + (axes.Count > 0 ? " [" + NativeShaderManifest.VariantKey(axes.Keys.OrderBy(k => k, StringComparer.Ordinal), axes) + "]" : ""); + return variant; + } + + // ------------------------------------------------------------------ interface extraction + + private static int LocationSpan(string type, int arrayLength) + { + Match matrix = Regex.Match(type, @"^d?mat(\d)(x\d)?$"); + int columns = matrix.Success ? int.Parse(matrix.Groups[1].Value, CultureInfo.InvariantCulture) : 1; + return columns * Math.Max(1, arrayLength); + } + + /// + /// The GLSL 330 stage's vertex inputs or fragment outputs after preprocessing, with the locations the + /// Vulkan path assigns: explicit ones first, then each unlocated declaration takes the lowest free span in + /// declaration order (ProgramInterfaceLayout.AssignInterfaceLocations; for a single unlocated output that + /// is location 0, as GL assigns it). + /// + internal static List<(int Location, string Name, string Type, int ArrayLength)> Glsl330Interface( + ShaderCompiler compiler, ShaderStageSource stage, GlslDeclarationKind kind) + { + ShaderCompileResult preprocessed = compiler.Preprocess(stage.Code, stage.PrefixCode, stage.Filename, stage.Stage); + Assert.True(preprocessed.Success, stage.Filename + ": " + preprocessed.Error); + + var declarations = GlslParser.Parse(preprocessed.PreprocessedText).Declarations.Where(d => d.Kind == kind).ToList(); + var used = new HashSet(); + var result = new List<(int, string, string, int)>(); + foreach (GlslDeclaration declaration in declarations.Where(d => d.Location >= 0)) + { + for (int i = 0; i < LocationSpan(declaration.TypeName, declaration.ArrayLength); i++) used.Add(declaration.Location + i); + result.Add((declaration.Location, declaration.Name, declaration.TypeName, declaration.ArrayLength)); + } + foreach (GlslDeclaration declaration in declarations.Where(d => d.Location < 0)) + { + int span = LocationSpan(declaration.TypeName, declaration.ArrayLength); + int location = 0; + while (Enumerable.Range(location, span).Any(used.Contains)) location++; + for (int i = 0; i < span; i++) used.Add(location + i); + result.Add((location, declaration.Name, declaration.TypeName, declaration.ArrayLength)); + } + return result.OrderBy(entry => entry.Item1).ToList(); + } + + private static string Describe(IEnumerable entries) => "[" + string.Join(", ", entries) + "]"; + + private static void CompareLists(List failures, string label, string what, IList native, IList glsl330) + { + if (native.SequenceEqual(glsl330)) return; + failures.Add(label + ": " + what + " differ: native " + Describe(native) + ", GLSL 330 " + Describe(glsl330) + + "; only native " + Describe(native.Except(glsl330)) + ", only GLSL 330 " + Describe(glsl330.Except(native))); + } + + /// The includes used by both stages of a program, transitively. + internal static SortedSet IncludesOf(string program) + { + var included = new SortedSet(StringComparer.Ordinal); + NativeShaderBuilder.ExpandIncludes(Path.Combine(SourceDirectory, program + ".glsl"), SourceDirectory, included); + return included; + } + + internal static List PortsOf(IEnumerable includes) => + includes.Where(name => File.Exists(Path.Combine(NativeShaderTree.IncludeDirectory, name))) + .Select(NativeShaderTree.PortOf).Where(port => port != null).Select(port => port!).ToList(); + + // ------------------------------------------------------------------ parity + + [SkippableFact] + public void TheNativeTreeBuildsWithoutErrors() + { + NativeShaderBuildResult result = RequireBuild(); + Assert.True(result.Success, string.Join("\n", result.Errors)); + Assert.NotEmpty(result.Manifest.Programs); + } + + /// + /// Section 2, per program and per variant: the name set with its GLSL types, the sampler order, the push + /// block limit, the include port headers' program uniforms, and - under every base variant - the vertex + /// inputs and fragment outputs. + /// + [SkippableTheory] + [MemberData(nameof(Programs))] + public void TheNativeProgramMatchesItsGlsl330Program(string program) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + NativeShaderBuildResult result = RequireBuild(); + List errors = ErrorsOf(result, program); + Assert.True(errors.Count == 0, string.Join("\n", errors)); + NativeProgram native = result.Manifest.FindProgram(program) + ?? throw new InvalidOperationException(program + " is missing from the manifest"); + + Dictionary files = ShaderCorpus.LoadShaderFiles(); + Dictionary includes = ShaderCorpus.LoadIncludes(); + Assert.True(files.ContainsKey(program + ".vsh") && files.ContainsKey(program + ".fsh"), + program + ": no GLSL 330 program " + program + ".vsh/.fsh to compare with (a native program is named after the GLSL 330 files it replaces)"); + + // The name set does not depend on defines: the oracle reads unpreprocessed text. + Oracle oracle = CollectOracle(ShaderCorpus.BuildProgram(program, files, includes, VariantFor("everything-off", new Dictionary()))); + List ports = PortsOf(IncludesOf(program)); + + Skip.IfNot(NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason), reason); + var failures = new List(); + using (compiler) + { + foreach (NativeVariant variant in native.Variants) + { + string label = program + (variant.Key.Length > 0 ? " [" + variant.Key + "]" : " [no axes]"); + CheckNames(failures, label, variant, oracle, ports); + CheckSamplerOrder(failures, label, variant, oracle); + CheckPortUniforms(failures, label, variant, ports); + if (variant.Push != null && variant.Push.Size > SetConvention.PushConstantBytes) + { + failures.Add(label + ": push block is " + variant.Push.Size + " B, the limit is " + SetConvention.PushConstantBytes); + } + + Dictionary axes = ParseKey(variant.Key); + foreach (string baseName in BaseVariants) + { + ShaderCorpus.ShaderVariant defines = VariantFor(baseName, axes); + List stages = ShaderCorpus.BuildProgram(program, files, includes, defines); + string where = label + " vs GLSL 330 " + defines.Name; + + var inputs = Glsl330Interface(compiler!, stages.Single(s => s.Stage == EnumShaderType.VertexShader), GlslDeclarationKind.Input); + CompareLists(failures, where, "vertex inputs (location type)", + variant.VertexInputs.OrderBy(v => v.Location).Select(v => v.Location + " " + v.Type + (v.ArrayLength != 0 ? "[" + v.ArrayLength + "]" : "")).ToList(), + inputs.Select(v => v.Location + " " + v.Type + (v.ArrayLength != 0 ? "[" + v.ArrayLength + "]" : "")).ToList()); + + var outputs = Glsl330Interface(compiler!, stages.Single(s => s.Stage == EnumShaderType.FragmentShader), GlslDeclarationKind.Output); + CompareLists(failures, where, "fragment outputs (location name)", + variant.FragmentOutputs.OrderBy(v => v.Location).Select(v => v.Location + " " + v.Name).ToList(), + outputs.Select(v => v.Location + " " + v.Name).ToList()); + } + } + } + + Assert.True(failures.Count == 0, string.Join("\n", failures)); + } + + /// Frame members through owner includes, frame textures of the ported includes, push and record members, sampler names. + private static void CheckNames(List failures, string label, NativeVariant variant, Oracle oracle, List ports) + { + var native = new SortedDictionary>(StringComparer.Ordinal); + void Add(string name, string type, string source) + { + if (!native.TryGetValue(name, out SortedSet? types)) native[name] = types = new SortedSet(StringComparer.Ordinal); + if (types.Count > 0 && !types.Contains(type)) failures.Add(label + ": '" + name + "' is declared twice with different types (" + string.Join("/", types) + " and " + type + " from " + source + ")"); + types.Add(type); + } + + foreach (string member in variant.FrameMembers) + { + Assert.True(FrameGlobals.TryGetMember(member, out UniformMember frame), member + " is not a FrameGlobals member"); + Add(member, frame.Type.Name, "the frame block"); + } + foreach (NativeShaderTree.Port port in ports) + { + foreach (NativeShaderTree.Declaration texture in port.FrameTextures) Add(texture.Name, texture.Type, port.Include); + } + var slots = variant.Samplers.Select(s => s.Name).ToHashSet(StringComparer.Ordinal); + foreach (NativeMember member in variant.Push?.Members ?? new List()) + { + if (!slots.Contains(member.Name)) Add(member.Name, member.Type, "the push block"); + } + foreach (NativeMember member in variant.Record?.Members ?? new List()) Add(member.Name, member.Type, "the record"); + foreach (NativeSampler sampler in variant.Samplers) Add(sampler.Name, sampler.GlslType, "a sampler slot"); + // A set-0 frame texture the program declares itself (cloudvolumetric's liquidDepth, without including + // underwatereffects) is the bindings.glsl declaration every native stage already has, as the rewriter + // treats a sampler whose name and type match SetConvention.FrameTextures. + foreach ((string name, SortedSet types) in oracle.Names) + { + if (native.ContainsKey(name)) continue; + foreach (string type in types) + { + if (SetConvention.FrameTextures.Any(binding => binding.Name == name && binding.GlslType == type)) + { + Add(name, type, "a set-0 frame texture the program declares itself"); + } + } + } + + var onlyNative = native.Keys.Except(oracle.Names.Keys).ToList(); + var onlyGlsl330 = oracle.Names.Keys.Except(native.Keys).ToList(); + var typeDiffers = native.Keys.Intersect(oracle.Names.Keys) + .Where(name => !native[name].SetEquals(oracle.Names[name])) + .Select(name => name + " native " + string.Join("/", native[name]) + " GLSL 330 " + string.Join("/", oracle.Names[name])) + .ToList(); + if (onlyNative.Count + onlyGlsl330.Count + typeDiffers.Count > 0) + { + failures.Add(label + ": uniform names differ from collectUniformNames: only native " + Describe(onlyNative) + + ", only GLSL 330 " + Describe(onlyGlsl330) + ", type differs " + Describe(typeDiffers)); + } + } + + /// The sampler order is the texture unit collectUniformNames assigns. + private static void CheckSamplerOrder(List failures, string label, NativeVariant variant, Oracle oracle) + { + CompareLists(failures, label, "samplers (texture unit name)", + variant.Samplers.OrderBy(s => s.Order).Select(s => s.Order + " " + s.Name).ToList(), + oracle.TextureLocations.Where(t => variant.Samplers.Any(s => s.Name == t.Key) || !IsFrameTexture(t.Key)) + .OrderBy(t => t.Value).Select((t, index) => index + " " + t.Key).ToList()); + + var duplicateUnits = oracle.TextureLocations.GroupBy(t => t.Value).Where(g => g.Count() > 1).Select(g => g.Key + ": " + string.Join("/", g.Select(t => t.Key))).ToList(); + if (duplicateUnits.Count > 0) failures.Add(label + ": the GLSL 330 program assigns one texture unit to several samplers " + Describe(duplicateUnits)); + } + + private static bool IsFrameTexture(string name) => SetConvention.FrameTextures.Any(binding => binding.Name == name); + + /// + /// Every program uniform an included port's header lists is a frame member when the program includes that + /// name's owner, and otherwise a push or record member with the header's type. + /// + private static void CheckPortUniforms(List failures, string label, NativeVariant variant, List ports) + { + var members = (variant.Push?.Members ?? new List()).Concat(variant.Record?.Members ?? new List()) + .ToDictionary(m => m.Name, m => m, StringComparer.Ordinal); + foreach (NativeShaderTree.Port port in ports) + { + foreach (NativeShaderTree.Declaration uniform in port.ProgramUniforms) + { + if (variant.FrameMembers.Contains(uniform.Name)) continue; + if (!members.TryGetValue(uniform.Name, out NativeMember? member)) + { + failures.Add(label + ": " + port.Include + " needs program uniform '" + uniform.Text + "' in the push block or record"); + continue; + } + string declared = member.Type + " " + member.Name + (member.ArrayLength > 0 ? "[" + member.ArrayLength + "]" : ""); + string wanted = Regex.Replace(uniform.Text, @"\[\s*([^\]]*?)\s*\]", m => + GlslParser.TryEvaluateConstantInt(m.Groups[1].Value, out int length) ? "[" + length + "]" : m.Value); + if (declared != wanted) failures.Add(label + ": " + port.Include + " needs '" + wanted + "', the program declares '" + declared + "'"); + } + } + } + + // ------------------------------------------------------------------ compile + + /// + /// Every variant of the program is built (2^axes of them), every shipped module passes spirv-val when it is on + /// PATH, and the push block and record reflect exactly as the source declares them: the same members in the + /// same order with scalar-layout offsets. + /// + [SkippableTheory] + [MemberData(nameof(Programs))] + public void EveryVariantCompilesValidatesAndReflectsItsDeclaredBlocks(string program) + { + NativeShaderBuildResult result = RequireBuild(); + List errors = ErrorsOf(result, program); + Assert.True(errors.Count == 0, string.Join("\n", errors)); + NativeProgram native = result.Manifest.FindProgram(program) + ?? throw new InvalidOperationException(program + " is missing from the manifest"); + Assert.Equal(1 << native.Axes.Count, native.Variants.Count); + + string? spirvVal = FindOnPath("spirv-val"); + var failures = new List(); + Skip.IfNot(NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason), reason); + using (compiler) + { + foreach (NativeVariant variant in native.Variants) + { + string label = program + (variant.Key.Length > 0 ? " [" + variant.Key + "]" : " [no axes]"); + Dictionary axes = ParseKey(variant.Key); + string prefix = string.Concat(axes.OrderBy(a => a.Key, StringComparer.Ordinal).Select(a => "#define " + a.Key + " " + a.Value + "\n")); + + Assert.Equal(new[] { "vertex", "fragment" }, variant.Stages.Select(s => s.Stage)); + foreach (NativeStage stage in variant.Stages) + { + if (spirvVal != null) Validate(failures, label, spirvVal, stage.Spirv, result.Files[stage.Spirv]); + + EnumShaderType type = stage.Stage == "vertex" ? EnumShaderType.VertexShader : EnumShaderType.FragmentShader; + string expanded = NativeShaderBuilder.ExpandIncludes(Path.Combine(SourceDirectory, stage.Source), SourceDirectory, new HashSet()); + string stagePrefix = "#define " + (type == EnumShaderType.VertexShader ? "OPTIMUM_VERTEX" : "OPTIMUM_FRAGMENT") + " 1\n"; + ShaderCompileResult preprocessed = compiler!.Preprocess(expanded, stagePrefix + prefix, stage.Source, type); + Assert.True(preprocessed.Success, label + " " + stage.Source + ": " + preprocessed.Error); + + CheckDeclaredBlock(failures, label + " " + stage.Source + " push block", DeclaredBlock(preprocessed.PreprocessedText, push: true), variant.Push); + CheckDeclaredBlock(failures, label + " " + stage.Source + " record", DeclaredBlock(preprocessed.PreprocessedText, push: false), variant.Record); + } + } + } + if (spirvVal == null) Console.WriteLine("spirv-val is not on PATH; the SPIR-V validation part of " + program + " was skipped"); + + Assert.True(failures.Count == 0, string.Join("\n", failures)); + } + + private static readonly Regex BlockDeclaration = new(@"layout\s*\(([^)]*)\)\s*uniform\s+(\w+)\s*\{([^}]*)\}", RegexOptions.Singleline); + private static readonly Regex MemberDeclaration = new(@"^\s*(\w+)\s+(\w+)\s*(?:\[\s*(\d+)\s*\])?\s*$"); + + /// The members (type, name, array length) of the push block or the record as the preprocessed source declares them, or null. + internal static List<(string Type, string Name, int ArrayLength)>? DeclaredBlock(string preprocessed, bool push) + { + foreach (Match block in BlockDeclaration.Matches(preprocessed)) + { + string layout = Regex.Replace(block.Groups[1].Value, @"\s+", ""); + bool isPush = layout.Split(',').Contains("push_constant"); + bool isRecord = layout.Split(',').Contains("set=" + SetConvention.StorageSet) && + layout.Split(',').Contains("binding=" + SetConvention.ProgramRecordBinding); + if (push ? !isPush : !isRecord) continue; + + var members = new List<(string, string, int)>(); + foreach (string statement in block.Groups[3].Value.Split(';')) + { + if (statement.Trim().Length == 0) continue; + Match member = MemberDeclaration.Match(statement); + Assert.True(member.Success, "cannot read block member '" + statement.Trim() + "'"); + members.Add((member.Groups[1].Value, member.Groups[2].Value, + member.Groups[3].Success ? int.Parse(member.Groups[3].Value, CultureInfo.InvariantCulture) : 0)); + } + return members; + } + return null; + } + + /// Bytes of one element under GL_EXT_scalar_block_layout, where every member aligns to its 4-byte scalar. + private static int ScalarSize(string type) + { + Match vector = Regex.Match(type, @"^[iub]?vec(\d)$"); + if (vector.Success) return 4 * int.Parse(vector.Groups[1].Value, CultureInfo.InvariantCulture); + Match matrix = Regex.Match(type, @"^mat(\d)(?:x(\d))?$"); + if (matrix.Success) + { + int columns = int.Parse(matrix.Groups[1].Value, CultureInfo.InvariantCulture); + int rows = matrix.Groups[2].Success ? int.Parse(matrix.Groups[2].Value, CultureInfo.InvariantCulture) : columns; + return 4 * columns * rows; + } + return type is "float" or "int" or "uint" or "bool" ? 4 : throw new InvalidOperationException("no scalar size for " + type); + } + + private static void CheckDeclaredBlock(List failures, string label, List<(string Type, string Name, int ArrayLength)>? declared, NativeBlock? reflected) + { + if (declared == null || reflected == null) + { + if ((declared == null) != (reflected == null)) failures.Add(label + ": declared " + (declared != null) + ", reflected " + (reflected != null)); + return; + } + + int offset = 0; + var expected = new List(); + foreach ((string type, string name, int arrayLength) in declared) + { + offset = (offset + 3) & ~3; + int size = ScalarSize(type) * Math.Max(1, arrayLength); + expected.Add(type + " " + name + (arrayLength != 0 ? "[" + arrayLength + "]" : "") + " @" + offset + " " + size); + offset += size; + } + CompareLists(failures, label, "members (type name @offset size)", + reflected.Members.Select(m => m.Type + " " + m.Name + (m.ArrayLength != 0 ? "[" + m.ArrayLength + "]" : "") + " @" + m.Offset + " " + m.Size).ToList(), + expected); + if (reflected.Size != offset) failures.Add(label + ": reflected size " + reflected.Size + " B, declared members end at " + offset + " B"); + } + + private static string? FindOnPath(string tool) + { + foreach (string directory in (Environment.GetEnvironmentVariable("PATH") ?? "").Split(Path.PathSeparator)) + { + if (directory.Length == 0) continue; + string candidate = Path.Combine(directory, OperatingSystem.IsWindows() ? tool + ".exe" : tool); + if (File.Exists(candidate)) return candidate; + } + return null; + } + + private static void Validate(List failures, string label, string spirvVal, string name, byte[] spirv) + { + string path = Path.Combine(Path.GetTempPath(), "optimum-native-parity-" + Guid.NewGuid().ToString("N") + ".spv"); + try + { + File.WriteAllBytes(path, spirv); + var start = new ProcessStartInfo(spirvVal) + { + RedirectStandardOutput = true, + RedirectStandardError = true, + UseShellExecute = false, + }; + // The environment the renderer creates: Vulkan 1.3 with scalarBlockLayout, which the device floor + // requires and the device enables (Core/VulkanContext.cs: the floor lists a missing scalarBlockLayout, + // device creation sets ScalarBlockLayout = true in the Vulkan 1.2 features). Every push block and + // record is scalar-laid-out (contract section 4), so without the flag + // spirv-val checks a device this renderer never runs on. + start.ArgumentList.Add("--target-env"); + start.ArgumentList.Add("vulkan1.3"); + start.ArgumentList.Add("--scalar-block-layout"); + start.ArgumentList.Add(path); + using Process process = Process.Start(start)!; + string output = process.StandardOutput.ReadToEnd() + process.StandardError.ReadToEnd(); + process.WaitForExit(); + if (process.ExitCode != 0) failures.Add(label + ": spirv-val rejects " + name + ":\n" + output); + } + finally + { + File.Delete(path); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeShaderRuntimeTests.cs b/Optimum.Render.Vulkan.Tests/NativeShaderRuntimeTests.cs new file mode 100644 index 00000000..d4f10033 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeShaderRuntimeTests.cs @@ -0,0 +1,691 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.IO; +using System.Linq; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; +using static Optimum.Render.Vulkan.Tests.GpuTest; +using TestProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using TestShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The native shader runtime seam (docs/vulkan.md): the manifest looked up at +/// VulkanDevice.LinkProgram by pass name and variant key, programs built from its SPIR-V with the +/// shared layout, specialization constants from the prefix, GLSL 330 initializers seeded, and a per-program +/// fall back to the rewriter. +/// +public sealed class NativeShaderRuntimeTests +{ + private const int Size = 16; + private static readonly string[] FamilyOne = { "blit", "final", "luma" }; + + private readonly ITestOutputHelper _output; + + public NativeShaderRuntimeTests(ITestOutputHelper output) => _output = output; + + // ------------------------------------------------------------------ variant keys + + private static NativeProgram AllAxes() => new() { Name = "every-axis", Axes = NativeShaderBuilder.VariantAxes.ToList() }; + + private static Dictionary DefinesOf(ShaderCorpus.ShaderVariant variant) => + NativeShaderLibrary.ParseDefines(new[] + { + ShaderCorpus.PrefixFor(EnumShaderType.VertexShader, variant), + ShaderCorpus.PrefixFor(EnumShaderType.FragmentShader, variant), + }); + + /// Every corpus variant keys as its settings say: GBUFFER from SSAOLEVEL > 0, the rest from their defines. + [Fact] + public void TheVariantKeyOfEveryCorpusVariantFollowsItsSettings() + { + foreach (ShaderCorpus.ShaderVariant variant in ShaderCorpus.Variants()) + { + string? key = NativeShaderLibrary.VariantKeyFor(AllAxes(), DefinesOf(variant), out string error); + Assert.True(key != null, variant.Name + ": " + error); + + var expected = new Dictionary + { + ["ALLOWDEPTHOFFSET"] = 0, + ["GBUFFER"] = variant.SsaoLevel > 0 ? 1 : 0, + ["GLOWSUB"] = 0, + ["GREEDYMESH"] = variant.GreedyMesh, + ["TAAMOTION"] = variant.TaaMotion, + ["USEOIT"] = variant.UseOit, + ["USESSBO"] = variant.UseSsbo, + ["VEC3SCALE"] = 0, + }; + Assert.Equal(NativeShaderManifest.VariantKey(expected.Keys, expected), key); + } + } + + /// + /// The runtime key is the inverse of the parity harness's mapping from a key back to ShaderRegistry's prefix: + /// every combination of every axis, on every harness base, comes back as the key it started from. + /// + [Fact] + public void EveryAxisCombinationRoundTripsThroughThePrefixOnEveryParityBase() + { + string[] axes = NativeShaderBuilder.VariantAxes; + foreach (string baseName in new[] { "everything-off", "everything-on", "taa-with-ssao" }) + { + for (int combination = 0; combination < 1 << axes.Length; combination++) + { + var values = new Dictionary(StringComparer.Ordinal); + for (int a = 0; a < axes.Length; a++) values[axes[a]] = (combination >> a) & 1; + string key = NativeShaderManifest.VariantKey(axes, values); + + ShaderCorpus.ShaderVariant variant = NativeShaderParityTests.VariantFor(baseName, values); + Assert.Equal(key, NativeShaderLibrary.VariantKeyFor(AllAxes(), DefinesOf(variant), out _)); + } + } + } + + /// Every variant the tree's manifest holds is the one the runtime asks for under the prefix that variant stands for. + [SkippableFact] + public void EveryManifestVariantKeyIsTheKeyItsPrefixProduces() + { + NativeShaderBuildResult build = RequireTreeBuild(); + int checkedKeys = 0; + foreach (NativeProgram program in build.Manifest.Programs) + { + foreach (NativeVariant variant in program.Variants) + { + foreach (string baseName in new[] { "everything-off", "everything-on", "taa-with-ssao" }) + { + ShaderCorpus.ShaderVariant corpus = NativeShaderParityTests.VariantFor(baseName, NativeShaderParityTests.ParseKey(variant.Key)); + string? key = NativeShaderLibrary.VariantKeyFor(program, DefinesOf(corpus), out string error); + Assert.True(key == variant.Key, program.Name + " [" + variant.Key + "] on " + baseName + ": got '" + key + "' " + error); + checkedKeys++; + } + } + } + Assert.True(checkedKeys > 0); + } + + [Fact] + public void AnAxisValueOutsideZeroAndOneHasNoKey() + { + var defines = NativeShaderLibrary.ParseDefines(new[] { "#define TAAMOTION 2\r\n" }); + Assert.Null(NativeShaderLibrary.VariantKeyFor(new NativeProgram { Axes = { "TAAMOTION" } }, defines, out string error)); + Assert.Contains("TAAMOTION", error); + } + + /// Each specialization constant takes the value ShaderRegistry stamps into the define it replaces. + [Fact] + public void SpecializationConstantsTakeTheirDefinesValues() + { + var variant = new NativeVariant(); + foreach (SpecializationConvention.Constant constant in SpecializationConvention.Constants) + { + variant.SpecializationConstants.Add(new NativeSpecConstant { Id = (int)constant.Id, Name = constant.Name, Type = constant.GlslType }); + } + + foreach (ShaderCorpus.ShaderVariant corpus in ShaderCorpus.Variants()) + { + Assert.True(NativeShaderLibrary.TryBuildSpecialization(variant, DefinesOf(corpus), out NativeSpecialization specialization, out string error), error); + var expected = new Dictionary + { + ["OPTIMUM_FXAA"] = corpus.Fxaa, ["OPTIMUM_SSAOLEVEL"] = corpus.SsaoLevel, ["OPTIMUM_NORMALVIEW"] = corpus.NormalView, + ["OPTIMUM_BLOOM"] = corpus.Bloom, ["OPTIMUM_GODRAYS"] = corpus.GodRays, ["OPTIMUM_FOAMEFFECT"] = corpus.FoamEffect, + ["OPTIMUM_SHINYEFFECT"] = corpus.ShinyEffect, ["OPTIMUM_SHADOWQUALITY"] = corpus.ShadowQuality, + ["OPTIMUM_WAVINGSTUFF"] = corpus.WavingStuff, ["OPTIMUM_MINBRIGHT"] = corpus.MinBright, + ["OPTIMUM_GREEDYMESH_GRAD"] = 0, ["OPTIMUM_DYNLIGHTS"] = corpus.DynLights, + ["OPTIMUM_OPTIMUMAO"] = corpus.OptimumAo, + }; + foreach (NativeSpecialization.Entry entry in specialization.Entries) + { + SpecializationConvention.Constant constant = SpecializationConvention.Constants.Single(c => c.Id == entry.Id); + Assert.Equal(4u, entry.Size); + double actual = constant.GlslType == "float" + ? BitConverter.ToSingle(specialization.Data, (int)entry.Offset) + : BitConverter.ToInt32(specialization.Data, (int)entry.Offset); + Assert.True(Math.Abs(expected[constant.Name] - actual) < 1e-6, corpus.Name + " " + constant.Name + ": " + actual); + } + Assert.Equal(SpecializationConvention.Constants.Length, specialization.Entries.Length); + } + } + + // ------------------------------------------------------------------ loading + + [Fact] + public void TheDisableVariableWinsOverEveryOtherSource() + { + Assert.Equal(NativeShaderLibrary.Mode.Off, NativeShaderLibrary.Resolve(null, "/dir", "0", "/src", "/bin").Mode); + Assert.Equal(NativeShaderLibrary.Mode.Off, NativeShaderLibrary.Resolve(false, "/dir", null, "/src", "/bin").Mode); + Assert.Equal((NativeShaderLibrary.Mode.Directory, "/dir"), Pick(NativeShaderLibrary.Resolve(null, "/dir", "1", "/src", "/bin"))); + Assert.Equal((NativeShaderLibrary.Mode.Source, "/src"), Pick(NativeShaderLibrary.Resolve(null, null, null, "/src", "/bin"))); + Assert.Equal((NativeShaderLibrary.Mode.Directory, Path.Combine("/bin", "shaders-vk")), + Pick(NativeShaderLibrary.Resolve(true, null, null, null, "/bin"))); + + Assert.Equal(NativeShaderLibrary.Mode.Directory, NativeShaderLibrary.Resolve(null, "/dir", "force", null, "/bin").Mode); + Assert.True(NativeShaderLibrary.IgnoresModScan("force")); + Assert.True(NativeShaderLibrary.IgnoresModScan(" FORCE ")); + Assert.False(NativeShaderLibrary.IgnoresModScan("0")); + Assert.False(NativeShaderLibrary.IgnoresModScan("1")); + Assert.False(NativeShaderLibrary.IgnoresModScan(null)); + + static (NativeShaderLibrary.Mode, string?) Pick((NativeShaderLibrary.Mode Mode, string? Path, string Reason) r) => (r.Mode, r.Path); + } + + [Fact] + public void AManifestThatCannotBeTrustedIsRejectedWithOneReason() + { + string directory = TemporaryDirectory(); + try + { + Assert.Null(NativeShaderLibrary.Load(directory, "tool", out string missing)); + Assert.Contains("no manifest", missing); + + var manifest = new NativeShaderManifest { Toolchain = "tool" }; + string path = Path.Combine(directory, NativeShaderManifest.FileName); + + File.WriteAllText(path, manifest.ToJson().Replace("\"schemaVersion\": 1", "\"schemaVersion\": 2")); + Assert.Null(NativeShaderLibrary.Load(directory, "tool", out string schema)); + Assert.Contains("schema version 2", schema); + + File.WriteAllText(path, "{ not json"); + Assert.Null(NativeShaderLibrary.Load(directory, "tool", out string malformed)); + Assert.Contains("rejected", malformed); + + File.WriteAllText(path, manifest.ToJson()); + Assert.Null(NativeShaderLibrary.Load(directory, "another tool", out string toolchain)); + Assert.Contains("toolchain", toolchain); + + Assert.NotNull(NativeShaderLibrary.Load(directory, "tool", out string accepted)); + Assert.Equal("", accepted); + } + finally + { + Directory.Delete(directory, true); + } + } + + /// + /// A manifest built with the same compile options by another shaderc build (another platform, another package) + /// is accepted: the per-stage SHA-256 pins the SPIR-V. The differing library is the note for the status line. + /// Different options are still refused. + /// + [Fact] + public void OnlyTheCompileOptionsOfTheToolchainMustMatch() + { + string directory = TemporaryDirectory(); + try + { + string options = ShaderCompiler.OptionsIdentity; + var manifest = new NativeShaderManifest { Toolchain = options + ";shaderc-sha256:linux" }; + File.WriteAllText(Path.Combine(directory, NativeShaderManifest.FileName), manifest.ToJson()); + + Assert.NotNull(NativeShaderLibrary.Load(directory, options + ";shaderc-sha256:linux", out string same)); + Assert.Equal("", same); + + Assert.NotNull(NativeShaderLibrary.Load(directory, options + ";silk-shaderc-2.23.0.0", out string otherLibrary)); + Assert.Contains("shaderc-sha256:linux", otherLibrary); + Assert.Contains("silk-shaderc-2.23.0.0", otherLibrary); + + Assert.Null(NativeShaderLibrary.Load(directory, options.Replace("performance", "size") + ";shaderc-sha256:linux", out string otherOptions)); + Assert.Contains("toolchain", otherOptions); + + Assert.Equal(("a;b", "c"), NativeShaderLibrary.SplitToolchain("a;b;c")); + Assert.Equal(("tool", ""), NativeShaderLibrary.SplitToolchain("tool")); + } + finally + { + Directory.Delete(directory, true); + } + } + + /// A SPIR-V file whose bytes are not the manifest's is refused, and stays refused. + [SkippableFact] + public void ASpirvFileWhoseHashDisagreesIsRefused() + { + NativeShaderBuildResult build = RequireFamilyOneBuild(); + string root = TemporaryDirectory(); + try + { + NativeShaderBuilder.Write(build, root); + string directory = Path.Combine(root, NativeShaderManifest.DirectoryName); + NativeShaderLibrary library = NativeShaderLibrary.Load(directory, build.Manifest.Toolchain, out string reason)!; + Assert.True(library != null, reason); + + NativeStage stage = library!.Manifest.Find("luma", "")!.Stages.Single(s => s.Stage == "fragment"); + CorruptOneByte(Path.Combine(directory, stage.Spirv)); + Assert.False(library.TryGetSpirv(stage, out _, out string error)); + Assert.Contains("sha256", error); + + NativeStage intact = library.Manifest.Find("blit", "")!.Stages.Single(s => s.Stage == "fragment"); + Assert.True(library.TryGetSpirv(intact, out byte[] bytes, out string intactError), intactError); + Assert.NotEmpty(bytes); + } + finally + { + Directory.Delete(root, true); + } + } + + [Fact] + public void TheStatsLineCarriesTheLoadCounts() => + Assert.Equal("stats.shaders native=3 rewritten=41 failed=1", + Core.VulkanStats.FormatShadersLine(3, 41, 1)); + + // ------------------------------------------------------------------ GPU + + /// A settings combination: the defines the post programs branch on. + private sealed record Settings(int Fxaa, int Bloom, int SsaoLevel, int GodRays) + { + public ShaderCorpus.ShaderVariant ToVariant() => new() + { + Name = ToString(), Fxaa = Fxaa, Bloom = Bloom, SsaoLevel = SsaoLevel, GodRays = GodRays, + TaaMotionLocation = SsaoLevel > 0 ? 4 : 2, + }; + } + + private static readonly Settings[] Combinations = + { + new(0, 0, 0, 0), + new(1, 1, 2, 1), + new(0, 1, 1, 0), + new(1, 0, 0, 2), + new(0, 0, 2, 1), + }; + + /// + /// blit, final and luma, linked natively and through the rewriter on one device, render the same pixels + /// for fixed inputs under several settings: exact for blit and luma, within 1/255 for final. final's + /// minlight/maxlight/minsat/maxsat and extraGamma are never set, so both paths rely on the seeded initializers + /// (a zero maxlight divides by zero in ColorGrade). + /// + [SkippableFact] + public void NativeAndRewrittenFamilyOneProgramsRenderTheSamePixels() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + NativeShaderBuildResult build = RequireFamilyOneBuild(); + string root = TemporaryDirectory(); + try + { + NativeShaderBuilder.Write(build, root); + Skip.IfNot(TryCreateDevice(_output, Path.Combine(root, NativeShaderManifest.DirectoryName), null, out VulkanDevice? device), + "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + _output.WriteLine("native shaders: " + seam.NativeShaderStatus); + int[] inputs = InputTextures(seam); + int pairs = 0; + + foreach (Settings settings in Combinations) + { + foreach (string name in FamilyOne) + { + int native = LinkGlsl330(seam, name, name, settings.ToVariant()); + int rewritten = LinkGlsl330(seam, name, name + "-rewriter", settings.ToVariant()); + Assert.True(seam.IsNativeProgram(native), name + " did not link natively: " + seam.GetError()); + Assert.False(seam.IsNativeProgram(rewritten)); + + if (name == "final") + { + byte[] record = seam.ProgramRecordForTests(native)!; + Assert.Equal(1f, BitConverter.ToSingle(record, seam.GetUniformLocation(native, "maxlight"))); + Assert.Equal(1f, BitConverter.ToSingle(record, seam.GetUniformLocation(native, "maxsat"))); + Assert.Equal(1f, BitConverter.ToSingle(record, seam.GetUniformLocation(native, "extraGamma"))); + Assert.Equal(0f, BitConverter.ToSingle(record, seam.GetUniformLocation(native, "minlight"))); + } + + byte[] nativePixels = Render(seam, native, name, inputs); + byte[] rewrittenPixels = Render(seam, rewritten, name, inputs); + int worst = WorstChannelDifference(nativePixels, rewrittenPixels); + _output.WriteLine(settings + " " + name + ": worst channel difference " + worst); + if (name == "final") + { + Assert.True(worst <= 1, settings + " final differs by " + worst + "/255"); + Assert.Contains(nativePixels, b => b != 0); + } + else + { + Assert.Equal(rewrittenPixels, nativePixels); + } + + seam.DeleteProgram(native); + seam.DeleteProgram(rewritten); + pairs++; + } + } + + Assert.Equal(Combinations.Length * FamilyOne.Length, pairs); + Assert.Equal((pairs, pairs, 0), seam.ShaderLinkCounts); + AssertClean(seam); + } + } + finally + { + Directory.Delete(root, true); + } + } + + /// A variant whose SPIR-V fails its hash links through the rewriter, counts as failed, and still draws. + [SkippableFact] + public void ACorruptedSpirvFileFallsBackToTheRewriterAndCountsAsFailed() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + NativeShaderBuildResult build = RequireFamilyOneBuild(); + string root = TemporaryDirectory(); + try + { + NativeShaderBuilder.Write(build, root); + string directory = Path.Combine(root, NativeShaderManifest.DirectoryName); + CorruptOneByte(Path.Combine(directory, build.Manifest.Find("final", "")!.Stages.Single(s => s.Stage == "fragment").Spirv)); + + Skip.IfNot(TryCreateDevice(_output, directory, null, out VulkanDevice? device), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + Settings settings = Combinations[1]; + int final = LinkGlsl330(seam, "final", "final", settings.ToVariant()); + int blit = LinkGlsl330(seam, "blit", "blit", settings.ToVariant()); + + Assert.False(seam.IsNativeProgram(final)); + Assert.True(seam.IsNativeProgram(blit)); + Assert.Equal((1, 0, 1), seam.ShaderLinkCounts); + + int[] inputs = InputTextures(seam); + Assert.Contains(Render(seam, final, "final", inputs), b => b != 0); + AssertClean(seam); + } + } + finally + { + Directory.Delete(root, true); + } + } + + /// OPTIMUM_VK_NATIVE_SHADERS=0 (the device setting it maps to) links every program through the rewriter. + [SkippableFact] + public void TurningNativeShadersOffLinksEveryProgramThroughTheRewriter() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + NativeShaderBuildResult build = RequireFamilyOneBuild(); + string root = TemporaryDirectory(); + try + { + NativeShaderBuilder.Write(build, root); + Assert.Equal(NativeShaderLibrary.Mode.Off, + NativeShaderLibrary.Resolve(null, Path.Combine(root, NativeShaderManifest.DirectoryName), "0", null, null).Mode); + + Skip.IfNot(TryCreateDevice(_output, Path.Combine(root, NativeShaderManifest.DirectoryName), false, out VulkanDevice? device), + "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + Assert.StartsWith("off", seam.NativeShaderStatus); + foreach (string name in FamilyOne) + { + Assert.False(seam.IsNativeProgram(LinkGlsl330(seam, name, name, Combinations[0].ToVariant()))); + } + Assert.Equal((0, FamilyOne.Length, 0), seam.ShaderLinkCounts); + AssertClean(seam); + } + } + finally + { + Directory.Delete(root, true); + } + } + + /// + /// The launcher's mod shader scan at the seam: a program it names links through the rewriter and counts as + /// rewritten while another links natively; a scan of "all" sends every program to the rewriter and says so in + /// the status line; OPTIMUM_VK_NATIVE_SHADERS=force (the device setting it maps to) ignores the scan. + /// + [SkippableFact] + public void AProgramTheModScanNamesLinksThroughTheRewriterUnlessForced() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped vanilla assets."); + NativeShaderBuildResult build = RequireFamilyOneBuild(); + string root = TemporaryDirectory(); + try + { + NativeShaderBuilder.Write(build, root); + string directory = Path.Combine(root, NativeShaderManifest.DirectoryName); + ShaderCorpus.ShaderVariant variant = Combinations[1].ToVariant(); + Func blitOnly = name => string.Equals(name, "blit", StringComparison.OrdinalIgnoreCase); + + Skip.IfNot(TryCreateDevice(_output, directory, null, out VulkanDevice? device, blitOnly, false), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + Assert.DoesNotContain("rewriter-only", seam.NativeShaderStatus); + int blit = LinkGlsl330(seam, "blit", "blit", variant); + int luma = LinkGlsl330(seam, "luma", "luma", variant); + int blitAgain = LinkGlsl330(seam, "blit", "blit", variant); + Assert.False(seam.IsNativeProgram(blit)); + Assert.False(seam.IsNativeProgram(blitAgain)); + Assert.True(seam.IsNativeProgram(luma), seam.GetError()); + Assert.Equal((1, 2, 0), seam.ShaderLinkCounts); + Assert.Contains(Render(seam, blit, "blit", InputTextures(seam)), b => b != 0); + AssertClean(seam); + } + + Skip.IfNot(TryCreateDevice(_output, directory, null, out device, _ => true, false), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + Assert.Contains("rewriter-only", seam.NativeShaderStatus); + foreach (string name in FamilyOne) + { + Assert.False(seam.IsNativeProgram(LinkGlsl330(seam, name, name, variant))); + } + Assert.Equal((0, FamilyOne.Length, 0), seam.ShaderLinkCounts); + AssertClean(seam); + } + + Skip.IfNot(TryCreateDevice(_output, directory, null, out device, _ => true, true), "No usable Vulkan device."); + using (device) + { + VulkanDevice seam = device!; + Assert.Contains("scan ignored", seam.NativeShaderStatus); + Assert.True(seam.IsNativeProgram(LinkGlsl330(seam, "blit", "blit", variant)), seam.GetError()); + Assert.Equal((1, 0, 0), seam.ShaderLinkCounts); + AssertClean(seam); + } + } + finally + { + Directory.Delete(root, true); + } + } + + // ------------------------------------------------------------------ helpers + + private static readonly Lazy<(NativeShaderBuildResult? Result, string Reason)> FamilyOneBuild = new(() => BuildTree(FamilyOne)); + private static readonly Lazy<(NativeShaderBuildResult? Result, string Reason)> TreeBuild = new(() => BuildTree(null)); + + private static string SourceDirectory => Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + + private static (NativeShaderBuildResult? Result, string Reason) BuildTree(string[]? programs) + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return (null, reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + if (programs == null) return (builder.Build(SourceDirectory), ""); + + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + foreach (string program in programs) + { + NativeShaderBuildResult one = builder.Build(SourceDirectory, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + return (merged, ""); + } + } + + private static NativeShaderBuildResult RequireFamilyOneBuild() => Require(FamilyOneBuild.Value); + + private static NativeShaderBuildResult RequireTreeBuild() => Require(TreeBuild.Value); + + private static NativeShaderBuildResult Require((NativeShaderBuildResult? Result, string Reason) built) + { + Skip.If(built.Result == null, built.Reason); + Assert.True(built.Result!.Success, string.Join("\n", built.Result.Errors)); + return built.Result; + } + + private static string TemporaryDirectory() + { + string directory = Path.Combine(Path.GetTempPath(), "optimum-native-runtime-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + return directory; + } + + private static void CorruptOneByte(string path) + { + byte[] bytes = File.ReadAllBytes(path); + bytes[bytes.Length / 2] ^= 0x5A; + File.WriteAllBytes(path, bytes); + } + + private static bool TryCreateDevice(ITestOutputHelper output, string manifestDirectory, bool? enabled, out VulkanDevice? device, + Func? modScan = null, bool? ignoreModScan = null) + { + VulkanDevice created = NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = enabled; + created.ShaderProgramOverriddenByMods = modScan; + created.IgnoreModShaderScan = ignoreModScan; + if (created.Initialize(IntPtr.Zero, 0, 0, out string failureReason)) + { + device = created; + return true; + } + output.WriteLine("Vulkan unavailable: " + failureReason); + created.Dispose(); + device = null; + return false; + } + + /// Links the GLSL 330 program as ShaderRegistry would, under . + private static int LinkGlsl330(VulkanDevice seam, string program, string passName, ShaderCorpus.ShaderVariant variant) + { + List stages = ShaderCorpus.BuildProgram(program, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + var linked = new TestProgram { PassName = passName }; + foreach (ShaderStageSource stage in stages) + { + var shader = new TestShader { Type = stage.Stage, Code = stage.Code, PrefixCode = stage.PrefixCode }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + return id; + } + + private static unsafe int Texture(VulkanDevice seam, Func pixel) + { + var pixels = new byte[Size * Size * 4]; + for (int y = 0; y < Size; y++) + { + for (int x = 0; x < Size; x++) + { + (byte r, byte g, byte b) = pixel(x, y); + int i = (y * Size + x) * 4; + pixels[i] = r; + pixels[i + 1] = g; + pixels[i + 2] = b; + pixels[i + 3] = 255; + } + } + fixed (byte* data = pixels) + { + return seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)data, false); + } + } + + /// Five distinct inputs, one per final sampler, with edges for FXAA to find. + private static int[] InputTextures(VulkanDevice seam) => new[] + { + Texture(seam, (x, y) => ((byte)(x * 16), (byte)(y * 16), (byte)(((x ^ y) & 1) * 200 + 20))), + Texture(seam, (x, y) => ((byte)((x + y) % 3 * 90), 0, 0)), + Texture(seam, (x, y) => (60, (byte)(x * 8 + 30), 120)), + Texture(seam, (x, y) => ((byte)(y * 10), (byte)(y * 5), (byte)(x * 3))), + Texture(seam, (x, y) => ((byte)(255 - x * 12), (byte)(200 - y * 6), 0)), + }; + + private static readonly string[] FinalSamplers = { "primaryScene", "glowParts", "bloomParts", "godrayParts", "ssaoScene" }; + + private static byte[] Render(VulkanDevice seam, int program, string name, int[] inputs) + { + int target = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, target, 0); + seam.SetDrawBuffers(framebuffer, 1); + + if (name == "final") + { + for (int i = 0; i < FinalSamplers.Length; i++) seam.SetSamplerUnit(program, FinalSamplers[i], i); + SetFloats(seam, program, "invFrameSizeIn", 1f / Size, 1f / Size); + SetFloats(seam, program, "sunPosScreenIn", 0.4f, 0.6f, 0.1f); + SetFloats(seam, program, "sunPos3dIn", 0.3f, 0.5f, 0.2f); + SetFloats(seam, program, "playerViewVector", 0.1f, 0.2f, 0.9f); + seam.SetUniform(program, seam.GetUniformLocation(program, "optimumSsaoInScene"), 0); + SetFloats(seam, program, "gammaLevel", 1.1f); + SetFloats(seam, program, "brightnessLevel", 0.95f); + SetFloats(seam, program, "contrastLevel", 0.05f); + SetFloats(seam, program, "sepiaLevel", 0.1f); + SetFloats(seam, program, "ambientBloomLevel", 0.4f); + SetFloats(seam, program, "damageVignetting", 0.3f); + SetFloats(seam, program, "damageVignettingSide", 0.2f); + SetFloats(seam, program, "frostVignetting", 0.25f); + SetFloats(seam, program, "windWaveCounter", 3f); + SetFloats(seam, program, "glitchEffectStrength", 0.35f); + } + else + { + seam.SetSamplerUnit(program, "scene", 0); + } + + seam.BeginFrame(); + for (int i = 0; i < inputs.Length; i++) seam.BindTexture(i, inputs[i]); + seam.BindFramebuffer(framebuffer); + seam.UseProgram(program); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawFullscreenTriangle(); + seam.Present(); + + byte[] pixels = seam.ReadBackLevel0ForTests(target); + seam.DeleteFramebuffer(framebuffer); + seam.DeleteTexture(target); + return pixels; + } + + private static void SetFloats(VulkanDevice seam, int program, string name, params float[] values) + { + int location = seam.GetUniformLocation(program, name); + Assert.True(location >= 0, name + " has no location"); + switch (values.Length) + { + case 1: seam.SetUniform(program, location, values[0]); break; + case 2: seam.SetUniform(program, location, values[0], values[1]); break; + case 3: seam.SetUniform(program, location, values[0], values[1], values[2]); break; + default: throw new ArgumentException(name); + } + } + + private static int WorstChannelDifference(byte[] a, byte[] b) + { + Assert.Equal(a.Length, b.Length); + int worst = 0; + for (int i = 0; i < a.Length; i++) worst = Math.Max(worst, Math.Abs(a[i] - b[i])); + return worst; + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeSkyTests.cs b/Optimum.Render.Vulkan.Tests/NativeSkyTests.cs new file mode 100644 index 00000000..9c8b74ea --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeSkyTests.cs @@ -0,0 +1,421 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The sky dome, drawn twice on one Vulkan device: through the seam's neutral body (the OpenGL +/// body's RenderMesh, the route every vanilla system still takes) and through the native pass +/// VulkanClientPlatform.RenderSkyDome records (docs/vulkan.md, decision 5 +/// stage 2 - the first world system on the native device API). +/// +/// Behavioural identity is the acceptance rule (decision 6): the same shader, the same mesh and +/// the same fixed state have to put the same pixels on Primary's scene and glow attachments, +/// and the native route must not touch the GL state tracker, a texture unit or a draw-buffer +/// mask while its pass is open. +/// +public class NativeSkyTests(ITestOutputHelper output) +{ + private const int Size = 16; + + /// The platform with no window: both routes take their size from this seam. + private sealed class SkyPlatform : VulkanClientPlatform + { + public SkyPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + // ------------------------------------------------------------------ the tests + + /// + /// The native sky pass draws what the seam's neutral body draws: one declared pass, one + /// native mesh draw, and the same scene and glow pixels. + /// + [SkippableFact] + public unsafe void TheNativeSkyPassMatchesTheSeamsNeutralBody() + { + using Session session = Open(); + + (byte[] statedScene, byte[] statedGlow) = session.RunFrame(native: false); + + long passesBefore = session.Seam.NativePassesForTests; + long meshDrawsBefore = session.Seam.NativeMeshDrawsForTests; + (byte[] nativeScene, byte[] nativeGlow) = session.RunFrame(native: true); + + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshDrawsBefore); + + output.WriteLine("scene centre stated " + Centre(statedScene) + " native " + Centre(nativeScene)); + Assert.Equal(statedScene, nativeScene); + Assert.Equal(statedGlow, nativeGlow); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The seam's neutral body draws through the generic stated route and the native route does not: + /// the switch is real, and "OFF is vanilla" holds for the route the OpenGL path takes. + /// + [SkippableFact] + public unsafe void TheNeutralBodyDrawsThroughTheStatedRouteAndTheNativeRouteDoesNot() + { + using Session session = Open(); + + long nativeDrawsBefore = session.Seam.NativeDrawsForTests; + long statedBefore = session.Platform.StatedDrawsForTests; + session.RunFrame(native: false); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeDrawsBefore); + Assert.True(session.Platform.StatedDrawsForTests - statedBefore > 0); + + session.RunFrame(native: true); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The pass is declared for the mesh's own vertex layout, so the pipeline it draws through + /// is a mesh pipeline and one per target, reused across frames rather than rebuilt. + /// + [SkippableFact] + public unsafe void TheSkyPassBuildsOneMeshPipelineAndKeepsIt() + { + using Session session = Open(); + + session.RunFrame(native: true); + int after = session.Seam.NativePipelinesForTests; + session.RunFrame(native: true); + session.RunFrame(native: true); + + Assert.Equal(after, session.Seam.NativePipelinesForTests); + Assert.Equal(3, session.Seam.NativeMeshDrawsForTests); + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- driving + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + private Session Open() + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + + Session? session = Session.TryOpen(output, manifest); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// + /// The platform, its device, the Primary target the Opaque stage binds, the sky program and + /// the dome mesh, installed the way the client installs them and put back afterwards. + /// + private sealed class Session : IDisposable + { + /// The model-view matrix the seam carries: identity, so the dome's clip positions stand. + private static readonly float[] Identity = + { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + + public SkyPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + public MeshRef Dome { get; private set; } = null!; + public int SkyTexture { get; private set; } + public int GlowTexture { get; private set; } + + private ShaderProgram sky = null!; + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + + public static unsafe Session? TryOpen(ITestOutputHelper output, string manifestDirectory) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-sky-" + Guid.NewGuid().ToString("N")); + var platform = new SkyPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Primary = CreatePrimary(seam); + InstallFrameBuffers(platform, session.Primary); + + var program = new ShaderProgram { PassName = "sky" }; + Link(seam, program, "sky", new[] { "projectionMatrix", "modelViewMatrix" }); + session.sky = program; + + session.SkyTexture = Gradient(seam, 0); + session.GlowTexture = Gradient(seam, 1); + string[] samplerNames = seam.SamplerNamesOf(program.ProgramId); + foreach (string name in samplerNames) + { + // The units the client's ShaderProgramSky setters bind: what the stated route + // resolves its samplers through. The native route passes the handles instead. + int unit = program.uniformLocations.Count + Array.IndexOf(samplerNames, name); + seam.SetSamplerUnit(program.ProgramId, name, unit); + seam.BindTexture(unit, name == "sky" ? session.SkyTexture + : name == "glow" ? session.GlowTexture : session.SkyTexture); + } + + session.Dome = platform.UploadMesh(BuildDome()); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + // The mesh goes first: VAO's finalizer reaches for ScreenManager.Platform, which is + // about to be the client's again, and a live handle there would crash the test host. + if (Dome != null) Platform.DeleteMesh(Dome); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// + /// One frame of the Opaque stage at the point the sky renderer runs: Primary bound and + /// cleared, depth test off, no culling, the program's uniforms set, then the seam. + /// + public unsafe (byte[] Scene, byte[] Glow) RunFrame(bool native) + { + VulkanDevice seam = Seam; + Platform.NativeSkyEnabled = native; + + Platform.BeginFrame(); + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, 0b11); + seam.ClearColor(0, 0.125f, 0.25f, 0.5f, 1f); + seam.ClearColor(1, 0.75f, 0.5f, 0.25f, 1f); + seam.ClearDepth(1f); + + Platform.CurrentFrameBuffer = Primary; + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + + seam.UseProgram(sky.ProgramId); + ShaderProgramBase.CurrentShaderProgram = sky; + seam.SetUniformMatrix(sky.ProgramId, sky.uniformLocations["projectionMatrix"], Identity); + seam.SetUniformMatrix(sky.ProgramId, sky.uniformLocations["modelViewMatrix"], Identity); + + Platform.RenderSkyDome(Dome, SkyTexture, GlowTexture, Identity); + + byte[] scene = Read(seam, Primary.ColorTextureIds[0]); + byte[] glow = Read(seam, Primary.ColorTextureIds[1]); + Platform.EndFrame(); + return (scene, glow); + } + + /// One attachment's pixels, read through a framebuffer that holds only it. + private unsafe byte[] Read(VulkanDevice seam, int texture) + { + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + seam.BindFramebuffer(Primary.FboId); + return pixels; + } + + // ----------------------------------------------------------------- fixtures + + /// Links one vanilla program as ShaderRegistry does and fills the locations the test sets. + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), new ShaderCorpus.ShaderVariant()); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// Primary as the Opaque stage has it: scene at 0, glow at 1, plus depth. + private static FrameBufferRef CreatePrimary(VulkanDevice seam) + { + var primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false), + }; + for (int slot = 0; slot < primary.ColorTextureIds.Length; slot++) + { + seam.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + primary.ColorTextureIds[slot], 0); + } + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + seam.SetDrawBuffers(primary.FboId, 0b11); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + private static void InstallFrameBuffers(SkyPlatform platform, FrameBufferRef primary) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + /// A small gradient, so a sampling difference between the routes would show. + private static unsafe int Gradient(VulkanDevice seam, int phase) + { + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + { + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30 + phase * 7); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = 255; + } + } + fixed (byte* first = pixels) + { + return seam.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + /// + /// The dome, as the sky program sees it: positions and a per-vertex colour, no UVs - + /// the shape SystemRenderSkyColor uploads (genIcosahedron with Uv nulled), reduced to + /// two triangles that cover the target so every pixel is comparable. + /// + private static MeshData BuildDome() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: false, withRgba: true, withFlags: false); + float[] positions = + { + -0.9f, -0.9f, 0.5f, + 0.9f, -0.9f, 0.5f, + 0.9f, 0.9f, 0.5f, + -0.9f, 0.9f, 0.5f, + }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertexSkipTex(positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + ColorUtil.WhiteArgb); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } + + // ---------------------------------------------------------------- native shaders + + /// The sky program's manifest, built once for the whole class. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + NativeShaderBuildResult one = builder.Build(source, "sky"); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-sky-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeSsaoChainTests.cs b/Optimum.Render.Vulkan.Tests/NativeSsaoChainTests.cs new file mode 100644 index 00000000..6839b3f2 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeSsaoChainTests.cs @@ -0,0 +1,827 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The post chain's ambient-occlusion step on the Vulkan platform: the vanilla SSAO pass, its +/// bilateral blur ping-pong and the AO composite that multiplies the visibility into Primary +/// colour 0 before the TAA resolve reads it. +/// +/// Acceptance is behavioural identity (docs/vulkan.md, decision 6): every +/// test here runs the same inputs through the OpenGL body - the lib virtual the chain switch falls +/// back to - and through the native route, and compares the pixels of all three written targets. +/// The settings that change this step are covered: SSAO quality 1 and 2 (one blur iteration or +/// three, and SSAOLEVEL 1 or 2 in the shaders), AO off, TAA off (no composite, no temporal +/// dither), a render scale below 1 and the GTAO mode with its own composite branch. Bloom, god +/// rays, FXAA and the AO debug view do not reach this step at all - they read what it leaves, and +/// the passes that consume it are covered where they live. +/// +public class NativeSsaoChainTests(ITestOutputHelper output) +{ + private const int Size = 16; + private const int HalfSize = Size / 2; + + private static readonly string[] Programs = { "ssao", "bilateralblur", "scene-ssao" }; + + // ------------------------------------------------------------------------ tests + + /// + /// Vanilla SSAO at both qualities with TAA running: the raw SSAO target, the blurred target + /// the final composition reads, and Primary colour 0 after the multiply all have to come out + /// of the native route exactly as the OpenGL body leaves them. Quality 1 runs the blur once, + /// quality 2 three times, and the shaders differ by SSAOLEVEL. + /// + [SkippableTheory] + [InlineData(1)] + [InlineData(2)] + public void TheVanillaSsaoStepMatchesTheOpenGlBody(int quality) + { + using Session session = Open(quality == 1 ? "ssao-only" : "taa-with-ssao", taa: quality != 1); + session.SsaoQuality = quality; + + Frame stated = session.Run(native: false); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + Frame nativeRoute = session.Run(native: true); + + // The raw pass, one blur half-iteration per pass, and the composite when TAA runs. + int blurPasses = quality == 1 ? 2 : 6; + int expected = 1 + blurPasses + (quality != 1 ? 1 : 0); + Assert.Equal(expected, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(expected, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(stated.Raw, nativeRoute.Raw); + Assert.Equal(stated.Blurred, nativeRoute.Blurred); + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.Equal(stated.SsaoInScene, nativeRoute.SsaoInScene); + + // The step really did something, or the comparison above would pass on two routes that + // both wrote nothing. + Assert.NotEqual(session.RawSeed, nativeRoute.Raw); + Assert.NotEqual(session.BlurredSeed, nativeRoute.Blurred); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The same step with TAA off: SSAO and its blur still run, the composite does not, and + /// Primary colour 0 comes back exactly as the frame seeded it on both routes - the AO is left + /// for the final composition to apply, which is what the OpenGL path has always done. + /// + [SkippableFact] + public void TheVanillaSsaoStepSkipsTheCompositeWithTaaOff() + { + using Session session = Open("ssao-only", taa: false); + + Frame stated = session.Run(native: false); + Frame nativeRoute = session.Run(native: true); + + Assert.Equal(stated.Raw, nativeRoute.Raw); + Assert.Equal(stated.Blurred, nativeRoute.Blurred); + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.Equal(session.SceneSeed, nativeRoute.Scene); + Assert.False(nativeRoute.SsaoInScene); + Assert.False(stated.SsaoInScene); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// A render scale below 1. The SSAO pass's screenSize is the body's + /// ssaaLevel * client * (ssaaLevel == 1 ? 0.5 : 1), which the dither's Bayer lattice is + /// laid out on, so it changes every occlusion value - except at exactly 0.5, where the + /// half-resolution fudge cancels and the value is the one render scale 1 produces. Both scales + /// are here: 0.5 because it is the shipped setting, and 0.75 because it is a scale where the + /// value really differs, which is what proves it reaches the pass at all. + /// + [SkippableFact] + public void TheVanillaSsaoStepMatchesTheOpenGlBodyBelowRenderScaleOne() + { + using Session session = Open("taa-with-ssao", taa: true); + + session.SsaaLevel = 0.5f; + Frame statedHalf = session.Run(native: false); + Frame nativeHalf = session.Run(native: true); + Assert.Equal(statedHalf.Raw, nativeHalf.Raw); + Assert.Equal(statedHalf.Blurred, nativeHalf.Blurred); + Assert.Equal(statedHalf.Scene, nativeHalf.Scene); + + session.SsaaLevel = 0.75f; + Frame stated = session.Run(native: false); + Frame nativeRoute = session.Run(native: true); + Assert.Equal(stated.Raw, nativeRoute.Raw); + Assert.Equal(stated.Blurred, nativeRoute.Blurred); + Assert.Equal(stated.Scene, nativeRoute.Scene); + + Assert.NotEqual(nativeHalf.Raw, nativeRoute.Raw); + Assert.NotEqual(statedHalf.Raw, stated.Raw); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// AO off (SSAO quality 0, so RenderSSAO is false): neither route runs a pass, neither target + /// moves, and the "AO is in the scene" flag stays false - which is what makes the final + /// composition apply nothing rather than multiply by an unwritten target. + /// + [SkippableFact] + public void TheAoStepDoesNothingWhenAmbientOcclusionIsOff() + { + using Session session = Open("ssao-only", taa: true); + session.RenderSsao = false; + + Frame stated = session.Run(native: false); + + long passesBefore = session.Seam.NativePassesForTests; + Frame nativeRoute = session.Run(native: true); + + Assert.Equal(0, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(stated.Raw, nativeRoute.Raw); + Assert.Equal(session.RawSeed, nativeRoute.Raw); + Assert.Equal(session.BlurredSeed, nativeRoute.Blurred); + Assert.Equal(session.SceneSeed, nativeRoute.Scene); + Assert.False(nativeRoute.SsaoInScene); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The GTAO mode: the platform's own visibility texture replaces vanilla SSAO, so the raw and + /// blurred targets are never written and the composite takes the OPTIMUMAO branch with + /// optimumAoMode = 1, sampling the G-buffer position and the OIT revealage for the attenuation + /// vanilla SSAO applies inside its own pass. The compute pass itself is not exercised here - + /// this is the raster step around it - so the visibility texture is supplied directly. + /// + [SkippableFact] + public void TheGtaoCompositeMatchesTheOpenGlBody() + { + using Session session = Open("taa-with-gtao", taa: true, gtao: true); + session.AmbientOcclusionTexture = session.GtaoVisibility; + + Frame stated = session.Run(native: false); + + long passesBefore = session.Seam.NativePassesForTests; + long drawsBefore = session.Seam.NativeDrawsForTests; + Frame nativeRoute = session.Run(native: true); + + // The composite alone: vanilla SSAO and its blur stood down. + Assert.Equal(1, session.Seam.NativePassesForTests - passesBefore); + Assert.Equal(1, session.Seam.NativeDrawsForTests - drawsBefore); + + Assert.Equal(stated.Scene, nativeRoute.Scene); + Assert.True(nativeRoute.SsaoInScene); + Assert.Equal(session.RawSeed, nativeRoute.Raw); + Assert.Equal(session.BlurredSeed, nativeRoute.Blurred); + Assert.NotEqual(session.SceneSeed, nativeRoute.Scene); + + GpuTest.AssertClean(session.Seam); + } + + /// + /// The step leaves the GL-shaped state the steps after it inherit exactly where the OpenGL + /// body leaves it: blending on, the depth test on, and the viewport back at full render + /// resolution - the Luma step sets no viewport of its own and would otherwise draw into the + /// SSAO target's half-resolution one. + /// + [SkippableFact] + public void TheNativeAoStepLeavesTheGlShapedStateWhereTheBodyLeavesIt() + { + using Session session = Open("taa-with-ssao", taa: true); + + session.Run(native: false); + (int Width, int Height) stated = session.Viewport; + session.Run(native: true); + + Assert.Equal(stated, session.Viewport); + Assert.Equal((Size, Size), session.Viewport); + + GpuTest.AssertClean(session.Seam); + } + + // ---------------------------------------------------------------------- session + + /// What one run of the step produced, on either route. + private readonly record struct Frame(byte[] Raw, byte[] Blurred, byte[] Scene, bool SsaoInScene); + + private Session Open(string variantName, bool taa, bool gtao = false) + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + Session? session = Session.TryOpen(output, manifest, variantName, taa, gtao); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + /// The Vulkan platform without a window, with the targets this step indexes. + private sealed class AoPlatform : VulkanClientPlatform + { + public AoPlatform() : base(null!) + { + } + + /// The visibility texture RenderOptimumAmbientOcclusion is made to return, or 0. + public int AmbientOcclusionTexture { get; set; } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + + public override int RenderOptimumAmbientOcclusion(float[] projectMatrix) => AmbientOcclusionTexture; + + /// Primary is the render resolution here, so the bind and the viewport are the base's. + public override void LoadFrameBuffer(EnumFrameBuffer framebuffer) + { + if (framebuffer == EnumFrameBuffer.Primary) + { + CurrentFrameBuffer = FrameBuffers[0]; + return; + } + base.LoadFrameBuffer(framebuffer); + } + } + + private sealed class Session : IDisposable + { + private const BindingFlags Hidden = BindingFlags.Instance | BindingFlags.NonPublic; + + public AoPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + + /// The three targets' contents as the frame seeds them, decoded the way a run decodes them. + public byte[] RawSeed { get; private set; } = Array.Empty(); + public byte[] BlurredSeed { get; private set; } = Array.Empty(); + public byte[] SceneSeed { get; private set; } = Array.Empty(); + + /// A prepared visibility texture, for the GTAO branch. + public int GtaoVisibility { get; private set; } + + public (int Width, int Height) Viewport { get; private set; } + + public int AmbientOcclusionTexture + { + set => Platform.AmbientOcclusionTexture = value; + } + + public int SsaoQuality + { + set => ClientSettings.SSAOQuality = value; + } + + public float SsaaLevel + { + set => typeof(ClientPlatformWindows).GetField("ssaaLevel", Hidden)!.SetValue(Platform, value); + } + + public bool RenderSsao + { + set => typeof(ClientPlatformWindows).GetField("RenderSSAO", Hidden)!.SetValue(Platform, value); + } + + private readonly List buffers = new(); + private FrameBufferRef primary = null!; + private FrameBufferRef transparent = null!; + private FrameBufferRef ssao = null!; + private FrameBufferRef blurHorizontal = null!; + private FrameBufferRef blurVertical = null!; + private float[] projection = null!; + + private int decodeProgram; + private int decodeTarget; + private int decodeFramebuffer; + + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + private ShaderProgramSsao? ssaoBefore; + private ShaderProgramBilateralblur? blurBefore; + private ShaderProgram? sceneSsaoBefore; + private bool taaBefore; + private bool gtaoBefore; + private int qualityBefore; + private DefaultShaderUniforms uniforms = new(); + + public static Session? TryOpen(ITestOutputHelper output, string manifestDirectory, + string variantName, bool taa, bool gtao) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-ssao-" + Guid.NewGuid().ToString("N")); + var platform = new AoPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + ssaoBefore = ShaderPrograms.Ssao, + blurBefore = ShaderPrograms.Bilateralblur, + sceneSsaoBefore = ShaderPrograms.SceneSsao, + taaBefore = OptimumConfig.Taa, + gtaoBefore = OptimumConfig.AmbientOcclusionShadersUseGtao, + qualityBefore = ClientSettings.SSAOQuality, + }; + ScreenManager.Platform = platform; + ScreenManager.FrameProfiler ??= new FrameProfilerUtil(static (string _) => { }); + platform.ShaderUniforms = session.uniforms; + + OptimumConfig.Taa = taa; + OptimumConfig.AmbientOcclusionShadersUseGtao = gtao; + ClientSettings.SSAOQuality = 2; + + session.BuildTargets(); + session.LinkPrograms(variantName); + session.InstallState(); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + ShaderPrograms.Ssao = ssaoBefore!; + ShaderPrograms.Bilateralblur = blurBefore!; + ShaderPrograms.SceneSsao = sceneSsaoBefore!; + OptimumConfig.Taa = taaBefore; + OptimumConfig.AmbientOcclusionShadersUseGtao = gtaoBefore; + ClientSettings.SSAOQuality = qualityBefore; + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + // ---------------------------------------------------------------- one run + + /// + /// One frame: the three targets seeded, the step run on the chosen route, everything read + /// back. The routes share the seeding and the readback, so a difference is the step's. + /// + public Frame Run(bool native) + { + Platform.NativePostChainEnabled = native; + Platform.BeginFrame(); + SeedFrame(); + Platform.RunPostStepAmbientOcclusionForTests(projection); + Viewport = ((int)Platform.stated.Viewport.Extent.Width, (int)Platform.stated.Viewport.Extent.Height); + var frame = new Frame( + Decode(ssao.ColorTextureIds[0]), + Decode(blurVertical.ColorTextureIds[0]), + Decode(primary.ColorTextureIds[0]), + Platform.OptimumPostSsaoInScene); + Platform.EndFrame(); + return frame; + } + + /// The state the world stages leave for the post chain, and the inputs both routes read. + private void SeedFrame() + { + VulkanDevice seam = Seam; + + seam.BindFramebuffer(transparent.FboId); + seam.SetDrawBuffers(transparent.FboId, 0b111); + seam.ClearColor(1, 0.8f, 0.8f, 0.8f, 1f); + + seam.BindFramebuffer(ssao.FboId); + seam.SetDrawBuffers(ssao.FboId, 0b1); + seam.ClearColor(0, 0.2f, 0.2f, 0.2f, 1f); + seam.BindFramebuffer(blurHorizontal.FboId); + seam.SetDrawBuffers(blurHorizontal.FboId, 0b1); + seam.ClearColor(0, 0.3f, 0.3f, 0.3f, 1f); + seam.BindFramebuffer(blurVertical.FboId); + seam.SetDrawBuffers(blurVertical.FboId, 0b1); + seam.ClearColor(0, 0.4f, 0.4f, 0.4f, 1f); + + seam.BindFramebuffer(primary.FboId); + seam.SetDrawBuffers(primary.FboId, 0b1111); + seam.ClearColor(0, 0.5f, 0.6f, 0.7f, 1f); + seam.ClearColor(1, 0.1f, 0.2f, 0.3f, 1f); + seam.ClearDepth(1f); + Platform.CurrentFrameBuffer = primary; + + seam.SetViewport(0, 0, Size, Size); + seam.SetBlend(true, EnumBlendMode.Standard); + seam.SetDepthTest(true); + seam.SetDepthMask(true); + seam.SetCullFace(false); + } + + // ---------------------------------------------------------------- readback + + /// + /// Any attachment through an RGBA8 copy, because the seam's readback is four bytes per + /// pixel from attachment 0. Both routes go through the same copy, so equal bytes here mean + /// equal texels: the decode cannot hide a difference it applies to both sides identically. + /// + private unsafe byte[] Decode(int textureId) + { + VulkanDevice seam = Seam; + seam.BindFramebuffer(decodeFramebuffer); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.UseProgram(decodeProgram); + seam.SetSamplerUnit(decodeProgram, "source", 15); + seam.BindTexture(15, textureId); + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.DrawFullscreenTriangle(); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.BindFramebuffer(decodeFramebuffer); + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + return pixels; + } + + // ----------------------------------------------------------------- fixture + + private void BuildTargets() + { + VulkanDevice seam = Seam; + + // Primary with the SSAO G-buffer: colour, glow, normal, position. The AO step never + // touches the motion attachment, so this target stops at the G-buffer. + primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + Texture(Size, EnumTextureInternalFormat.Rgba8), + Texture(Size, EnumTextureInternalFormat.Rgba8), + GBuffer(normals: true), + GBuffer(normals: false), + }, + DepthTextureId = seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.DepthComponent32, + EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false), + }; + for (int slot = 0; slot < 4; slot++) + { + seam.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + primary.ColorTextureIds[slot], 0); + } + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string primaryStatus), primaryStatus); + + transparent = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + Texture(Size, EnumTextureInternalFormat.Rgba8), + Texture(Size, EnumTextureInternalFormat.Rgba8), + Texture(Size, EnumTextureInternalFormat.Rgba8), + }, + }; + for (int slot = 0; slot < 3; slot++) + { + seam.AttachTexture(transparent.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + transparent.ColorTextureIds[slot], 0); + } + Assert.True(seam.CheckFramebufferComplete(transparent.FboId, out string status), status); + + // The SSAO target as the platform builds it: half resolution, an RGB float attachment + // and the 16x16 rotation noise, which is a texture of the target rather than an + // attachment of it. + ssao = new FrameBufferRef + { + Width = HalfSize, + Height = HalfSize, + FboId = seam.CreateFramebuffer(HalfSize, HalfSize), + ColorTextureIds = new int[2], + }; + ssao.ColorTextureIds[0] = seam.CreateTexture2DRaw(HalfSize, HalfSize, 6407, IntPtr.Zero, 12); + seam.AttachTexture(ssao.FboId, EnumFramebufferAttachment.ColorAttachment0, ssao.ColorTextureIds[0], 0); + seam.SetDrawBuffers(ssao.FboId, 0b1); + ssao.ColorTextureIds[1] = Noise(seam); + + blurVertical = Blur(seam); + blurHorizontal = Blur(seam); + + GtaoVisibility = Visibility(seam); + + for (int i = 0; i <= 24; i++) buffers.Add(null!); + buffers[0] = primary; + buffers[1] = transparent; + buffers[13] = ssao; + buffers[14] = blurVertical; + buffers[15] = blurHorizontal; + + decodeTarget = Texture(Size, EnumTextureInternalFormat.Rgba8); + decodeFramebuffer = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(decodeFramebuffer, EnumFramebufferAttachment.ColorAttachment0, decodeTarget, 0); + seam.SetDrawBuffers(decodeFramebuffer, 0b1); + + projection = Mat4f.Perspective(Mat4f.Create(), 70f * (float)Math.PI / 180f, 1f, 0.1f, 100f); + } + + private int Texture(int size, EnumTextureInternalFormat format) => + Seam.CreateTexture2D(size, size, format, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + + private FrameBufferRef Blur(VulkanDevice seam) + { + var target = new FrameBufferRef + { + Width = HalfSize, + Height = HalfSize, + FboId = seam.CreateFramebuffer(HalfSize, HalfSize), + ColorTextureIds = new[] { Texture(HalfSize, EnumTextureInternalFormat.Rgba8) }, + }; + seam.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + seam.SetDrawBuffers(target.FboId, 0b1); + return target; + } + + /// The SSAO rotation noise, built by the platform's own generator so the pattern is the shipped one. + private static unsafe int Noise(VulkanDevice seam) + { + float[] noise = VulkanClientPlatform.BuildOptimumSsaoNoise(new Random(5), 16); + int id; + fixed (float* data = noise) id = seam.CreateTexture2DRaw(16, 16, 34836, (IntPtr)data, 16); + seam.SetTextureParameter(id, OptimumGlConstants.TextureWrapS, OptimumGlConstants.Repeat); + seam.SetTextureParameter(id, OptimumGlConstants.TextureWrapT, OptimumGlConstants.Repeat); + return id; + } + + /// + /// A G-buffer attachment with a pattern the SSAO kernel actually responds to: view-space + /// positions on a slope for the position target, and unit normals for the normal one. + /// + private unsafe int GBuffer(bool normals) + { + var texels = new float[Size * Size * 4]; + for (int y = 0; y < Size; y++) + { + for (int x = 0; x < Size; x++) + { + int i = (y * Size + x) * 4; + if (normals) + { + var normal = new Vec3f(0.1f + x * 0.01f, 0.15f, 1f); + normal.Normalize(); + texels[i] = normal.X; + texels[i + 1] = normal.Y; + texels[i + 2] = normal.Z; + // w is vanilla's leaves flag; 0 keeps the ordinary occlusion branch. + texels[i + 3] = 0f; + } + else + { + // Alternating depth columns: vanilla's SSAO clamps every kernel tap to + // within 0.04 of the fragment's own texcoord, so occlusion only comes from + // a near neighbour, and a fine step is what every pixel can see one of. + texels[i] = (x - Size * 0.5f) * 0.12f + 0.011f; + texels[i + 1] = (y - Size * 0.5f) * 0.12f; + texels[i + 2] = -3f + ((x * 7 + y * 13) % 11) * 0.02f; + texels[i + 3] = 0.1f; + } + } + } + fixed (float* data = texels) + { + return Seam.CreateTexture2DRaw(Size, Size, 34836, (IntPtr)data, 16); + } + } + + /// A visibility texture standing in for GTAO's output: R32F, nearest, clamped, as the platform sets it up. + private unsafe int Visibility(VulkanDevice seam) + { + var texels = new float[Size * Size]; + for (int i = 0; i < texels.Length; i++) texels[i] = 0.25f + (i % 7) / 12f; + int id; + fixed (float* data = texels) id = seam.CreateTexture2DRaw(Size, Size, 0x822E, (IntPtr)data, 4); + // Composed with texelFetch at the same resolution; nearest and clamp keep any sampling exact. + seam.SetTextureParameter(id, OptimumGlConstants.TextureMinFilter, 9728); + seam.SetTextureParameter(id, OptimumGlConstants.TextureMagFilter, 9728); + seam.SetTextureParameter(id, OptimumGlConstants.TextureWrapS, OptimumGlConstants.ClampToEdge); + seam.SetTextureParameter(id, OptimumGlConstants.TextureWrapT, OptimumGlConstants.ClampToEdge); + return id; + } + + private void LinkPrograms(string variantName) + { + VulkanDevice seam = Seam; + ShaderCorpus.ShaderVariant variant = Variant(variantName); + + var ssaoProgram = new ShaderProgramSsao { PassName = "ssao" }; + Link(seam, ssaoProgram, "ssao", variant, + new[] { "screenSize", "projection", "samples" }, + new[] { "gPosition", "gNormal", "texNoise", "revealage" }, + variant.TaaMotion == 1 ? new[] { "temporalFrameIndex" } : Array.Empty()); + var blurProgram = new ShaderProgramBilateralblur { PassName = "bilateralblur" }; + Link(seam, blurProgram, "bilateralblur", variant, new[] { "frameSize", "isVertical" }, + new[] { "inputTexture", "depthTexture" }, Array.Empty()); + var composite = new ShaderProgram { PassName = "scene-ssao" }; + Link(seam, composite, "scene-ssao", variant, new[] { "invRenderHeight" }, + variant.OptimumAo == 1 + ? new[] { "ssaoScene", "gPositionScene", "revealageScene" } + : new[] { "ssaoScene" }, + variant.OptimumAo == 1 ? new[] { "optimumAoMode" } : Array.Empty()); + + ShaderPrograms.Ssao = ssaoProgram; + ShaderPrograms.Bilateralblur = blurProgram; + ShaderPrograms.SceneSsao = composite; + + decodeProgram = LinkDecode(seam); + } + + internal static ShaderCorpus.ShaderVariant Variant(string name) + { + foreach (ShaderCorpus.ShaderVariant candidate in ShaderCorpus.Variants()) + { + if (candidate.Name == name) return candidate; + } + throw new InvalidOperationException("the corpus has no " + name + " variant"); + } + + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, + ShaderCorpus.ShaderVariant variant, string[] uniforms, string[] samplers, string[] optional) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), variant); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader), seam.GetError() ?? name + " did not compile"); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + // The GL route binds samplers by name through the program's location table. + foreach (string sampler in samplers) + { + int location = seam.GetUniformLocation(id, sampler); + Assert.True(location != -1, name + " has no location for " + sampler); + program.uniformLocations[sampler] = location; + } + // Uniforms the shader only declares in some variants: registered where they exist, so + // both routes leave them alone in the variants that do not have them. + foreach (string uniform in optional) + { + int location = seam.GetUniformLocation(id, uniform); + if (location != -1) program.uniformLocations[uniform] = location; + } + } + + /// The readback helper's own program: any attachment into RGBA8. + private static int LinkDecode(VulkanDevice seam) + { + const string vertex = @"#version 330 core +out vec2 uv; +void main(void) +{ + vec2 position = vec2((gl_VertexID << 1) & 2, gl_VertexID & 2); + uv = position; + gl_Position = vec4(position * 2.0 - 1.0, 0.0, 1.0); +} +"; + const string fragment = @"#version 330 core +uniform sampler2D source; +in vec2 uv; +layout(location = 0) out vec4 outColor; +void main(void) +{ + outColor = clamp(texture(source, uv), 0.0, 1.0); +} +"; + var linked = new LinkedProgram { PassName = "native-ssao-decode" }; + var vertexShader = new LinkedShader { Type = EnumShaderType.VertexShader, Code = vertex, PrefixCode = "" }; + var fragmentShader = new LinkedShader { Type = EnumShaderType.FragmentShader, Code = fragment, PrefixCode = "" }; + Assert.True(seam.CompileShader(vertexShader), seam.GetError() ?? "decode vertex shader"); + Assert.True(seam.CompileShader(fragmentShader), seam.GetError() ?? "decode fragment shader"); + linked.VertexShader = vertexShader; + linked.FragmentShader = fragmentShader; + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "decode link failed"); + return id; + } + + private void InstallState() + { + typeof(ClientPlatformWindows).GetField("frameBuffers", Hidden)!.SetValue(Platform, buffers); + typeof(ClientPlatformWindows).GetField("ssaaLevel", Hidden)!.SetValue(Platform, 1f); + typeof(ClientPlatformWindows).GetField("RenderSSAO", Hidden)!.SetValue(Platform, true); + // The composite only runs while TAA is actually accumulating, which is what the + // OptimumTaaRequested && TaaTargetsReady guard says. + typeof(ClientPlatformWindows).GetField("optimumTaaTargetsReady", Hidden)! + .SetValue(Platform, OptimumConfig.Taa); + FillSsaoKernel(); + + Platform.BeginFrame(); + SeedFrame(); + RawSeed = Decode(ssao.ColorTextureIds[0]); + BlurredSeed = Decode(blurVertical.ColorTextureIds[0]); + SceneSeed = Decode(primary.ColorTextureIds[0]); + Platform.EndFrame(); + } + + /// The 64-sample kernel, built the way the platform's frame-buffer setup builds it. + private void FillSsaoKernel() + { + float[] kernel = Platform.OptimumSsaoKernel; + var random = new Random(11); + for (int sample = 0; sample < 64; sample++) + { + var value = new Vec3f((float)random.NextDouble() * 2f - 1f, + (float)random.NextDouble() * 2f - 1f, (float)random.NextDouble()); + value.Normalize(); + value *= (float)random.NextDouble(); + float scale = sample / 64f; + scale = GameMath.Lerp(0.1f, 1f, scale * scale); + value *= scale; + kernel[sample * 3] = value.X; + kernel[sample * 3 + 1] = value.Y; + kernel[sample * 3 + 2] = value.Z; + } + } + } + + // ---------------------------------------------------------------- native shaders + + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + foreach (string program in Programs) + { + NativeShaderBuildResult one = builder.Build(source, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-ssao-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/NativeWorldSystemsTests.cs b/Optimum.Render.Vulkan.Tests/NativeWorldSystemsTests.cs new file mode 100644 index 00000000..7a54a0c7 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/NativeWorldSystemsTests.cs @@ -0,0 +1,823 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The world systems Phase 3b stage 2 moved onto the native device API after the sky dome - the +/// night sky box, the moon, the cube particle pool and the decal pool - each drawn twice on one +/// Vulkan device: through its seam's neutral body (the OpenGL body's own draw, the route every +/// system that has not moved still takes) and through the native pass the Vulkan platform +/// records (docs/vulkan.md, decision 5 stage 2). +/// +/// Behavioural identity is the acceptance rule (decision 6): the same program, the same mesh and +/// the same fixed state have to put the same pixels on every attachment of Primary - the motion +/// attachment included, bit for bit, because a world pass that writes motion has to land exactly +/// what the GL path lands there - and the native route must not touch the GL state tracker, a +/// texture unit or a draw-buffer mask while its pass is open. +/// +/// The sky dome itself is pinned by NativeSkyTests and the device's mesh-draw entry points by +/// NativeMeshDrawTests; those files are not duplicated here. +/// +public class NativeWorldSystemsTests(ITestOutputHelper output) +{ + private const int Size = 16; + + /// Primary's colour slots in this fixture: scene, glow, then the motion attachment. + private const int SceneSlot = 0; + private const int GlowSlot = 1; + private const int MotionSlot = 2; + + /// The platform with no window: both routes take their size from this seam. + private sealed class WorldPlatform : VulkanClientPlatform + { + public WorldPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + // ------------------------------------------------------------------------- night sky + + /// + /// The star box's native pass draws what its seam's neutral body draws: one declared pass, + /// one native mesh draw of the cube through the samplerCube the program declares, and the same pixels on every attachment. + /// + [SkippableFact] + public void TheNativeNightSkyPassMatchesTheSeamsNeutralBody() + { + using Session session = Open("nightsky"); + int cube = session.CubeGradient(); + + byte[][] stated = session.RunFrame(native: false, blending: false, depth: false, motion: false, + s => s.Platform.RenderNightSkyBox(s.Mesh, cube)); + + long passes = session.Seam.NativePassesForTests; + long meshes = session.Seam.NativeMeshDrawsForTests; + byte[][] native = session.RunFrame(native: true, blending: false, depth: false, motion: false, + s => s.Platform.RenderNightSkyBox(s.Mesh, cube)); + + Assert.Equal(1, session.Seam.NativePassesForTests - passes); + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshes); + AssertSameAttachments(stated, native, "nightsky"); + GpuTest.AssertClean(session.Seam); + } + + // -------------------------------------------------------------------------- celestial + + /// + /// The moon's native pass matches its seam's neutral body, with the body texture resolved + /// from its handle and the sky and glow frame textures resolved from theirs rather than from + /// the units ShaderProgramCelestialobject's setters bound them to. + /// + [SkippableFact] + public void TheNativeCelestialPassMatchesTheSeamsNeutralBody() + { + using Session session = Open("celestialobject"); + int body = session.Gradient(0); + int sky = session.Gradient(1); + int glow = session.Gradient(2); + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: false, motion: false, + s => s.Platform.RenderCelestialQuad(s.Mesh, body, sky, glow)); + + long meshes = session.Seam.NativeMeshDrawsForTests; + byte[][] native = session.RunFrame(native: true, blending: true, depth: false, motion: false, + s => s.Platform.RenderCelestialQuad(s.Mesh, body, sky, glow)); + + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshes); + AssertSameAttachments(stated, native, "celestialobject"); + GpuTest.AssertClean(session.Seam); + } + + // -------------------------------------------------------------------------- sun + + /// + /// The sun's native pass matches its seam's neutral body: standard's samplers resolved from the + /// pipeline's own declaration, the sun texture from its handle, blended with no depth test. + /// + [SkippableFact] + public void TheSunMatchesTheSeamsNeutralBody() + { + using Session session = Open("standard"); + int sun = session.Gradient(0); + + void Draw(Session s) + { + // What SystemRenderSunMoon writes, reduced to what makes the quad land: lit white, + // untinted, identity transforms, a low alpha test. + float[] identity = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + s.Program.UniformMatrix("modelMatrix", identity); + s.Program.UniformMatrix("viewMatrix", identity); + s.Program.Uniform("rgbaTint", 1f, 1f, 1f, 1f); + s.Program.Uniform("rgbaLightIn", 1f, 1f, 1f, 1f); + s.Program.Uniform("rgbaAmbientIn", 1f, 1f, 1f); + s.Program.Uniform("alphaTest", 0.01f); + // The fixture's frame block is zero, and the global warp reads it; the route under + // test does not depend on the warp, so the fixture skips it. + s.Program.Uniform("dontWarpVertices", 1); + s.Platform.BindProgramTexture2D(s.Program, "tex", sun, 0); + s.Platform.RenderSunQuad(s.Mesh, sun); + } + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: false, motion: false, Draw); + + long meshes = session.Seam.NativeMeshDrawsForTests; + byte[][] native = session.RunFrame(native: true, blending: true, depth: false, motion: false, Draw); + + Assert.Equal(1, session.Seam.NativeMeshDrawsForTests - meshes); + AssertSameAttachments(stated, native, "standard"); + GpuTest.AssertClean(session.Seam); + } + + /// A mod program registered under "standard" is not the vanilla one and stays on the neutral body. + [SkippableFact] + public void AModStandardProgramStaysOnTheNeutralBody() + { + using Session session = Open("standard"); + int sun = session.Gradient(0); + ShaderProgramStandard registered = ShaderPrograms.Standard; + ShaderPrograms.Standard = new ShaderProgramStandard(); + try + { + long meshes = session.Seam.NativeMeshDrawsForTests; + session.RunFrame(native: true, blending: true, depth: false, motion: false, + s => s.Platform.RenderSunQuad(s.Mesh, sun)); + Assert.Equal(0, session.Seam.NativeMeshDrawsForTests - meshes); + } + finally + { + ShaderPrograms.Standard = registered; + } + GpuTest.AssertClean(session.Seam); + } + + // -------------------------------------------------------------------------- particles + + /// + /// The cube pool's native pass matches its seam's neutral body, and the draw is recorded as + /// an instanced draw rather than as as many single draws. + /// Known gap: the fixture's quad carries no per-instance attributes, so the cubes do not land + /// on the scene slot and the pixel comparison is between two untouched attachments. What this + /// pins is the route and the draw kind, not the pixels. + /// + [SkippableFact] + public void TheNativeParticlePassMatchesTheSeamsNeutralBody() + { + using Session session = Open("particlescube"); + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: true, motion: false, + s => s.Platform.RenderParticles(s.Mesh, 4, 0)); + + long instanced = session.Seam.NativeInstancedDrawsForTests; + byte[][] native = session.RunFrame(native: true, blending: true, depth: true, motion: false, + s => s.Platform.RenderParticles(s.Mesh, 4, 0)); + + Assert.Equal(1, session.Seam.NativeInstancedDrawsForTests - instanced); + AssertSameAttachments(stated, native, "particlescube", mustDraw: false); + GpuTest.AssertClean(session.Seam); + } + + /// + /// Inside a motion window the cube pool's native pass takes the motion attachment into its + /// colour slots, and every attachment - the motion one bit for bit - comes out of the native + /// route exactly as it comes out of the neutral body's draw under the same window. This is + /// the temporal contract: a world pass that writes motion lands what the GL path lands. + /// + [SkippableFact] + public void TheNativeParticlePassLeavesTheMotionAttachmentIdentical() + { + using Session session = Open("particlescube"); + session.OpenMotionWindow(); + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: true, motion: true, + s => s.Platform.RenderParticles(s.Mesh, 4, 0)); + byte[][] native = session.RunFrame(native: true, blending: true, depth: true, motion: true, + s => s.Platform.RenderParticles(s.Mesh, 4, 0)); + + Assert.Equal(stated[MotionSlot], native[MotionSlot]); + AssertSameAttachments(stated, native, "particlescube (motion window)", mustDraw: false); + GpuTest.AssertClean(session.Seam); + } + + // ----------------------------------------------------------------------------- decals + + /// + /// The decal pool's native pass matches its seam's neutral body, and the draw is recorded as + /// one indirect multi-draw out of the per-slot indirect ring rather than one draw per group. + /// + [SkippableFact] + public void TheNativeDecalPassMatchesTheSeamsNeutralBody() + { + using Session session = Open("decals"); + int decal = session.Gradient(0); + int block = session.Gradient(1); + int[] starts = { 0, 0, 3 * 4, 0 }; + int[] sizes = { 3, 3 }; + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: true, motion: false, + s => + { + // The lib's route: the scope opens, vanilla MeshDataPool.Draw's own RenderMesh + // multi-draw runs inside it, the scope closes. + // What ShaderProgramDecals' setters do before the scope: the atlases on units. + s.Platform.BindProgramTexture2D(s.Program, "blockTexture", block, 0); + s.Platform.BindProgramTexture2D(s.Program, "decalTexture", decal, 1); + s.Platform.BeginDecalPass(decal, block); + try + { + s.Platform.RenderMesh(s.Mesh, starts, sizes, 2, false); + } + finally + { + s.Platform.EndDecalPass(); + } + }); + + long indirect = session.Seam.NativeIndirectDrawsForTests; + byte[][] native = session.RunFrame(native: true, blending: true, depth: true, motion: false, + s => + { + // The lib's route: the scope opens, vanilla MeshDataPool.Draw's own RenderMesh + // multi-draw runs inside it, the scope closes. + // What ShaderProgramDecals' setters do before the scope: the atlases on units. + s.Platform.BindProgramTexture2D(s.Program, "blockTexture", block, 0); + s.Platform.BindProgramTexture2D(s.Program, "decalTexture", decal, 1); + s.Platform.BeginDecalPass(decal, block); + try + { + s.Platform.RenderMesh(s.Mesh, starts, sizes, 2, false); + } + finally + { + s.Platform.EndDecalPass(); + } + }); + + Assert.Equal(1, session.Seam.NativeIndirectDrawsForTests - indirect); + AssertSameAttachments(stated, native, "decals"); + GpuTest.AssertClean(session.Seam); + } + + /// + /// Inside a motion window the decal pass's motion attachment is identical between the two + /// routes, for the same reason the particle pass's is: a decal writes the motion vector of + /// the surface it sits on, with its own nudged depth. + /// + [SkippableFact] + public void TheNativeDecalPassLeavesTheMotionAttachmentIdentical() + { + using Session session = Open("decals"); + session.OpenMotionWindow(); + int decal = session.Gradient(0); + int block = session.Gradient(1); + int[] starts = { 0, 0, 3 * 4, 0 }; + int[] sizes = { 3, 3 }; + + byte[][] stated = session.RunFrame(native: false, blending: true, depth: true, motion: true, + s => + { + // The lib's route: the scope opens, vanilla MeshDataPool.Draw's own RenderMesh + // multi-draw runs inside it, the scope closes. + // What ShaderProgramDecals' setters do before the scope: the atlases on units. + s.Platform.BindProgramTexture2D(s.Program, "blockTexture", block, 0); + s.Platform.BindProgramTexture2D(s.Program, "decalTexture", decal, 1); + s.Platform.BeginDecalPass(decal, block); + try + { + s.Platform.RenderMesh(s.Mesh, starts, sizes, 2, false); + } + finally + { + s.Platform.EndDecalPass(); + } + }); + byte[][] native = session.RunFrame(native: true, blending: true, depth: true, motion: true, + s => + { + // The lib's route: the scope opens, vanilla MeshDataPool.Draw's own RenderMesh + // multi-draw runs inside it, the scope closes. + // What ShaderProgramDecals' setters do before the scope: the atlases on units. + s.Platform.BindProgramTexture2D(s.Program, "blockTexture", block, 0); + s.Platform.BindProgramTexture2D(s.Program, "decalTexture", decal, 1); + s.Platform.BeginDecalPass(decal, block); + try + { + s.Platform.RenderMesh(s.Mesh, starts, sizes, 2, false); + } + finally + { + s.Platform.EndDecalPass(); + } + }); + + Assert.Equal(stated[MotionSlot], native[MotionSlot]); + AssertSameAttachments(stated, native, "decals (motion window)"); + GpuTest.AssertClean(session.Seam); + } + + // ------------------------------------------------------------------- switch and slots + + /// + /// The neutral body draws through the generic stated route and the native route does not: the + /// switch is real, and "OFF is vanilla" holds for the route the OpenGL path takes. + /// + [SkippableFact] + public void TheNeutralBodiesDrawThroughTheStatedRouteAndTheNativeRouteDoesNot() + { + using Session session = Open("particlescube"); + + long nativeBefore = session.Seam.NativeDrawsForTests; + long statedBefore = session.Platform.StatedDrawsForTests; + session.RunFrame(native: false, blending: true, depth: true, motion: false, + s => s.Platform.RenderParticles(s.Mesh, 2, 0)); + Assert.Equal(0, session.Seam.NativeDrawsForTests - nativeBefore); + Assert.True(session.Platform.StatedDrawsForTests - statedBefore > 0); + + session.RunFrame(native: true, blending: true, depth: true, motion: false, + s => s.Platform.RenderParticles(s.Mesh, 2, 0)); + GpuTest.AssertClean(session.Seam); + } + + /// + /// The colour slots a native world pass declares are the set the stated route's + /// draw-buffer mask holds at the same point in the frame, derived from the platform's own + /// motion-window state: Primary's default colour set, plus the motion attachment exactly + /// while a window is open, and every bound slot with TAA off. + /// + [SkippableFact] + public void TheDeclaredColourSlotsAreTheOnesTheStatedMaskWouldHold() + { + using Session session = Open("particlescube"); + MethodInfo slots = typeof(VulkanClientPlatform).GetMethod("NativeWorldPassColorSlots", + BindingFlags.Instance | BindingFlags.NonPublic)!; + + // TAA off: the attachment index is negative and the pass takes every bound slot. + session.Platform.SetOptimumMotionAttachmentIndex(-1); + Assert.Equal(0b111u, (uint)slots.Invoke(session.Platform, new object[] { session.Primary })!); + + // TAA on, window closed: Primary's default colour set, the motion attachment out. + session.Platform.SetOptimumMotionAttachmentIndex(MotionSlot); + SetMotionWriteActive(session.Platform, false); + Assert.Equal(0b011u, (uint)slots.Invoke(session.Platform, new object[] { session.Primary })!); + + // TAA on, window open: the motion attachment joins the set. + SetMotionWriteActive(session.Platform, true); + Assert.Equal(0b111u, (uint)slots.Invoke(session.Platform, new object[] { session.Primary })!); + } + + // ---------------------------------------------------------------------------- helpers + + /// The scene slot's centre as RunFrame clears it (0.125, 0.25, 0.5). + private const string ClearedSceneCentre = "32,64,127,255"; + + private void AssertSameAttachments(byte[][] stated, byte[][] native, string what, bool mustDraw = true) + { + for (int slot = 0; slot < stated.Length; slot++) + { + output.WriteLine(what + " slot " + slot + " centre stated " + Centre(stated[slot]) + + " native " + Centre(native[slot])); + Assert.Equal(stated[slot], native[slot]); + } + // Two untouched attachments are equal too. Until the fixture seeded the frame block and + // the transforms, every comparison in this file was exactly that. + if (mustDraw) Assert.NotEqual(ClearedSceneCentre, Centre(native[SceneSlot])); + } + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + /// + /// The window flag ClientPlatformWindows keeps private. The tests set it directly rather + /// than through BeginMotionWrite, which also wants a temporal frame, a jitter window and the + /// TAA targets - none of which change what is under test here. + /// + private static void SetMotionWriteActive(VulkanClientPlatform platform, bool active) => + typeof(ClientPlatformWindows) + .GetField("optimumMotionWriteActive", BindingFlags.Instance | BindingFlags.NonPublic)! + .SetValue(platform, active); + + private Session Open(string program) + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + + Session? session = Session.TryOpen(output, manifest, program); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + // ---------------------------------------------------------------------------- driving + + /// + /// The platform, its device, the Primary target its stage binds (scene, glow and the motion + /// attachment), one vanilla program and one mesh, installed the way the client installs them + /// and put back afterwards. + /// + private sealed class Session : IDisposable + { + private static readonly float[] Identity = + { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + + public WorldPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public FrameBufferRef Primary { get; private set; } = null!; + public MeshRef Mesh { get; private set; } = null!; + + private ShaderProgram program = null!; + private ClientPlatformAbstract? previousPlatform; + private ShaderProgramStandard? previousStandard; + private bool registeredStandard; + + /// The linked program, for a test that sets uniforms of its own. + public ShaderProgram Program => program; + private string dataPath = ""; + private int gradients; + + public static unsafe Session? TryOpen(ITestOutputHelper output, string manifestDirectory, string programName) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-world-" + Guid.NewGuid().ToString("N")); + var platform = new WorldPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + previousPlatform = ScreenManager.Platform, + dataPath = dataPath, + }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.Primary = CreatePrimary(seam); + InstallFrameBuffers(platform, session.Primary); + // TAA on with the motion attachment appended after Primary's default colour set, so + // the window-derived slot masks under test mean something. + platform.SetOptimumMotionAttachmentIndex(MotionSlot); + + // standard is the one vanilla program the native route checks by identity (a mod can + // register its own under the same name), so it is built as the vanilla type and + // registered where the client registers it. + bool standard = programName == "standard"; + ShaderProgram linked = standard + ? new ShaderProgramStandard { PassName = programName } + : new ShaderProgram { PassName = programName }; + Link(seam, linked, programName, standard + ? new[] { "projectionMatrix", "modelMatrix", "viewMatrix", "rgbaTint", "rgbaLightIn", "rgbaAmbientIn", "alphaTest", "dontWarpVertices" } + : new[] { "projectionMatrix" }); + session.program = linked; + if (standard) + { + session.previousStandard = ShaderPrograms.Standard; + ShaderPrograms.Standard = (ShaderProgramStandard)linked; + session.registeredStandard = true; + } + session.Mesh = platform.UploadMesh(BuildQuad()); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + if (registeredStandard) ShaderPrograms.Standard = previousStandard!; + // The mesh goes first: VAO's finalizer reaches for ScreenManager.Platform, which is + // about to be the client's again, and a live handle there would crash the test host. + if (Mesh != null) Platform.DeleteMesh(Mesh); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// Opens the caller's motion window, the way BeginMotionWrite leaves the platform. + public void OpenMotionWindow() => SetMotionWriteActive(Platform, true); + + /// + /// One frame at the point the system under test runs: Primary bound and cleared, the + /// draw-buffer mask the motion window would have left, the caller's blend, depth and + /// cull state, the program in use, then the seam. + /// + public unsafe byte[][] RunFrame(bool native, bool blending, bool depth, bool motion, Action draw) + { + VulkanDevice seam = Seam; + Platform.NativeWorldEnabled = native; + uint mask = motion ? 0b111u : 0b011u; + + Platform.BeginFrame(); + seam.BindFramebuffer(Primary.FboId); + seam.SetDrawBuffers(Primary.FboId, 0b111); + seam.ClearColor(SceneSlot, 0.125f, 0.25f, 0.5f, 1f); + seam.ClearColor(GlowSlot, 0.75f, 0.5f, 0.25f, 1f); + seam.ClearColor(MotionSlot, 0.375f, 0.625f, 0.875f, 1f); + seam.ClearDepth(1f); + seam.SetDrawBuffers(Primary.FboId, (int)mask); + + Platform.CurrentFrameBuffer = Primary; + seam.SetViewport(0, 0, Size, Size); + seam.SetDepthTest(depth); + seam.SetDepthMask(depth); + seam.SetCullFace(false); + seam.SetBlend(blending, EnumBlendMode.Standard); + // The replace-blending the window forces on the motion attachment, which the native + // pass states per attachment instead. + if (motion) Platform.ApplyOptimumMotionBlendState(); + + seam.UseProgram(program.ProgramId); + ShaderProgramBase.CurrentShaderProgram = program; + seam.SetUniformMatrix(program.ProgramId, program.uniformLocations["projectionMatrix"], Identity); + SeedFrameGlobals(seam, program.ProgramId); + SeedDrawUniforms(seam, program.ProgramId); + + draw(this); + + var pixels = new byte[Primary.ColorTextureIds.Length][]; + for (int slot = 0; slot < pixels.Length; slot++) pixels[slot] = Read(seam, Primary.ColorTextureIds[slot]); + Platform.EndFrame(); + return pixels; + } + + /// + /// The frame globals ShaderProgramBase.Use() would have written. RunFrame binds the program + /// directly, so without these the shared frame block stays zero - and a zero viewDistance + /// makes standard.vsh's distance fade a division by zero that discards every fragment, which + /// is how these comparisons were once equal without anything having been drawn. + /// + private static void SeedFrameGlobals(VulkanDevice seam, int programId) + { + foreach ((string name, float value) in new[] + { + ("zNear", 0.1f), ("zFar", 1000f), ("viewDistance", 1000f), ("viewDistanceLod0", 1000f), + }) + { + int location = seam.GetUniformLocation(programId, name); + if (location != -1) seam.SetUniform(programId, location, value); + } + } + + /// + /// Identity transforms and white light for whichever of them the program declares, so the + /// quad lands on the scene slot. Left at zero, the matrices collapse every vertex to one + /// point and the comparison is between two untouched attachments. + /// + private static void SeedDrawUniforms(VulkanDevice seam, int programId) + { + foreach (string matrix in new[] { "modelMatrix", "viewMatrix", "modelViewMatrix" }) + { + int location = seam.GetUniformLocation(programId, matrix); + if (location != -1) seam.SetUniformMatrix(programId, location, Identity); + } + foreach (string colour in new[] { "rgbaAmbientIn", "rgbaLightIn", "rgbaTint" }) + { + int location = seam.GetUniformLocation(programId, colour); + if (location != -1) seam.SetUniform(programId, location, 1f, 1f, 1f, 1f); + } + } + + /// One attachment's pixels, read through a framebuffer that holds only it. + private unsafe byte[] Read(VulkanDevice seam, int texture) + { + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + seam.BindFramebuffer(Primary.FboId); + return pixels; + } + + // ----------------------------------------------------------------- fixtures + + /// A small 2D gradient, so a sampling difference between the routes would show. + public unsafe int Gradient(int phase) + { + gradients++; + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + { + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30 + phase * 7); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = 255; + } + } + fixed (byte* first = pixels) + { + return Seam.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + /// + /// The star cube map: six faces, one gradient each, uploaded the way + /// ClientPlatformWindows.Load3DTextureCube uploads SystemRenderNightSky's stars. The + /// native pass has to resolve it into the bindless table's cube array, not the 2D one. + /// + public unsafe int CubeGradient() + { + const int face = 8; + var faces = new byte[6][]; + var pointers = new IntPtr[6]; + var handles = new System.Runtime.InteropServices.GCHandle[6]; + for (int f = 0; f < 6; f++) + { + faces[f] = new byte[face * face * 4]; + for (int y = 0; y < face; y++) + { + for (int x = 0; x < face; x++) + { + int i = (y * face + x) * 4; + faces[f][i] = (byte)(f * 40); + faces[f][i + 1] = (byte)(x * 30); + faces[f][i + 2] = (byte)(y * 30); + faces[f][i + 3] = 255; + } + } + handles[f] = System.Runtime.InteropServices.GCHandle.Alloc( + faces[f], System.Runtime.InteropServices.GCHandleType.Pinned); + pointers[f] = handles[f].AddrOfPinnedObject(); + } + + try + { + return Seam.CreateTextureCube(face, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, pointers); + } + finally + { + for (int f = 0; f < 6; f++) handles[f].Free(); + } + } + + /// Links one vanilla program as ShaderRegistry does and fills the locations the test sets. + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), new ShaderCorpus.ShaderVariant()); + + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode, + }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + /// Primary as a world stage has it: scene at 0, glow at 1, motion at 2, plus depth. + private static FrameBufferRef CreatePrimary(VulkanDevice seam) + { + var primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new int[3], + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false), + }; + for (int slot = 0; slot < primary.ColorTextureIds.Length; slot++) + { + primary.ColorTextureIds[slot] = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + seam.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + primary.ColorTextureIds[slot], 0); + } + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + seam.SetDrawBuffers(primary.FboId, 0b111); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + private static void InstallFrameBuffers(WorldPlatform platform, FrameBufferRef primary) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + /// + /// Two triangles covering the target, with positions, UVs, a colour and flags - enough + /// for every program under test, whose remaining vertex inputs the layout fills with the + /// constant defaults GL promises. Six indices, so a multi-draw can take them as two + /// groups of three. + /// + private static MeshData BuildQuad() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + float[] positions = + { + -0.9f, -0.9f, 0.5f, + 0.9f, -0.9f, 0.5f, + 0.9f, 0.9f, 0.5f, + -0.9f, 0.9f, 0.5f, + }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertex(positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } + + // ---------------------------------------------------------------- native shaders + + /// The four programs' manifest, built once for the whole class. + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + var merged = new NativeShaderBuildResult(); + merged.Manifest.Toolchain = compiler!.Identity; + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + foreach (string program in new[] { "nightsky", "celestialobject", "particlescube", "decals", "standard" }) + { + NativeShaderBuildResult one = builder.Build(source, program); + merged.Errors.AddRange(one.Errors); + merged.Manifest.Programs.AddRange(one.Manifest.Programs); + foreach ((string file, byte[] bytes) in one.Files) merged.Files[file] = bytes; + } + if (!merged.Success) return ("", string.Join("\n", merged.Errors)); + + string root = Path.Combine(Path.GetTempPath(), "optimum-native-world-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(merged, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/Optimum.Render.Vulkan.Tests.csproj b/Optimum.Render.Vulkan.Tests/Optimum.Render.Vulkan.Tests.csproj new file mode 100644 index 00000000..1879b4e0 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Optimum.Render.Vulkan.Tests.csproj @@ -0,0 +1,36 @@ + + + + net10.0 + false + true + annotations + + + + + + + + + + + + + + + + + + + + + ..\.vanilla\win-x64\vintagestory\VintagestoryAPI.dll + true + + + + diff --git a/Optimum.Render.Vulkan.Tests/PipelineCacheTests.cs b/Optimum.Render.Vulkan.Tests/PipelineCacheTests.cs new file mode 100644 index 00000000..eb2deafa --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/PipelineCacheTests.cs @@ -0,0 +1,276 @@ +using System; +using System.Collections.Generic; +using System.Diagnostics; +using System.Globalization; +using System.IO; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class PipelineCacheTests(ITestOutputHelper output) +{ + private const string Vertex = """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); } + """; + + [SkippableFact] + public void GamePipelinesReuseStateAndPersistForTheSameDriver() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + var messages = new List(); + var context = GpuTest.CreateContext(output, messages); + string root = Path.Combine(Path.GetTempPath(), "optimum-cache-" + Guid.NewGuid().ToString("N")); + try + { + using var compiler = new ShaderCompiler(); + var identity = PipelineCacheIdentity.Of(context.Capabilities); + var persistence = PipelineCachePersistence.Open(root, identity, out var seed, + thresholdBytes: 1, interval: TimeSpan.FromSeconds(1)); + Assert.Null(seed); + using var cache = new GraphicsPipelineCache(context) { KeyLog = persistence.KeyLog }; + var files = ShaderCorpus.LoadShaderFiles(); var includes = ShaderCorpus.LoadIncludes(); + var variant = ShaderCorpus.Variants().Single(v => v.Name == "everything-on"); + string[] names = { "blit", "final", "luma", "findbright", "godrays", "chunkopaque" }; + var programs = new List(); + try + { + for (int i = 0; i < names.Length; i++) + { + var translated = ShaderTranslator.Translate(ShaderCorpus.BuildProgram(names[i], files, includes, variant), compiler); + Assert.True(translated.Success, names[i] + ": " + string.Join("; ", translated.Errors)); + var program = new ShaderProgramResources(context, i + 1, translated); + programs.Add(program); + Assert.NotEqual(0ul, program.PipelineLayout.Handle); + var shared = program.StandaloneLayout!; + foreach (var layout in new[] { shared.FrameSetLayout, shared.TextureSetLayout, shared.StorageSetLayout }) + Assert.NotEqual(0ul, layout.Handle); + if (names[i] == "chunkopaque") + { + var storage = Assert.Single(translated.Layout.StorageBlocks); + Assert.Equal(SetConvention.StorageSet, storage.Set); + Assert.Equal(SetConvention.FaceDataBinding, storage.Binding); + continue; // Its vertex and face-data drawing path is covered by terrain acceptance. + } + if (names[i] == "final") + { + Assert.True(translated.Layout.Samplers.Count >= 4); + Assert.True(translated.Layout.BlockSize > 0); + } + var targets = new RenderTargetFormats(new[] { Format.R8G8B8A8Unorm }, Format.Undefined); + var key = new PipelineKey(i + 1, 0, 0, 0, PolygonMode.Fill, 2); + AttachmentBlend blend = AttachmentBlend.Default; + GraphicsPipelineCache.PipelineRequest Request() => new() { + Program = program, VertexLayout = VertexLayoutDescription.Empty, Targets = targets, + Blend = new[] { blend }, PolygonMode = PolygonMode.Fill, Topology = PrimitiveTopology.TriangleList, + }; + var first = cache.Get(key, Request()); + Assert.NotEqual(0ul, first.Handle); + Assert.Equal(first.Handle, cache.Get(key, Request()).Handle); + Assert.Equal(first.Handle, cache.Get(key, Request()).Handle); + blend = AttachmentBlend.For(true, EnumBlendMode.Glow); + key = key with { BlendId = 1 }; + Assert.NotEqual(first.Handle, cache.Get(key, Request()).Handle); + } + Assert.Equal(10, cache.Count); Assert.Equal(10, cache.Misses); Assert.Equal(10, cache.Hits); + long second = Stopwatch.Frequency; + Assert.False(persistence.Tick(cache, 10 * second)); + Assert.False(persistence.Tick(cache, 10 * second + second / 2)); + Assert.True(persistence.Tick(cache, 11 * second)); persistence.WaitForPendingSave(); + Assert.Equal(1, persistence.Saves); Assert.False(persistence.KeyLog.HasUnsavedChanges); + Assert.Equal(10, PipelineKeyLog.Load(persistence.KeyLogPath).Count); + var saved = PipelineCacheFile.Load(persistence.CachePath, identity); + Assert.NotNull(saved); Assert.True(PipelineCacheFile.HasMatchingVulkanHeader(saved, identity)); + using (var next = new GraphicsPipelineCache(context, saved)) Assert.True(next.SeedAccepted); + persistence.SaveAtShutdown(cache); Assert.Equal(2, persistence.Saves); + } + finally { foreach (var program in programs) program.Dispose(); } + } + finally { context.Dispose(); DeleteRoot(root); } + ValidationAssert.NoErrors(messages); + } + + [Fact] + public void TheGrowthTriggerSamplesAtMostOncePerIntervalAndFiresOnThreshold() + { + const long mib = 1024 * 1024; + long second = Stopwatch.Frequency; + var trigger = new PipelineCacheGrowthTrigger(8 * mib, TimeSpan.FromSeconds(10), baselineBytes: 2 * mib); + + // The first call arms the clock; then one sample per interval, never two. + Assert.False(trigger.SampleDue(100 * second)); + Assert.False(trigger.SampleDue(105 * second)); + Assert.True(trigger.SampleDue(110 * second)); + Assert.False(trigger.SampleDue(119 * second)); + Assert.True(trigger.SampleDue(121 * second)); + + // Growth is measured from what is on disk: the seed first, then the last save. + Assert.False(trigger.GrewEnough(9 * mib)); + Assert.True(trigger.GrewEnough(10 * mib)); + trigger.NoteSaved(10 * mib); + Assert.Equal(10 * mib, trigger.BaselineBytes); + Assert.False(trigger.GrewEnough(17 * mib)); + Assert.True(trigger.GrewEnough(18 * mib)); + } + + [Fact] + public void SynchronousPipelinesComeFromTheSettingOrTheEnvironment() + { + Assert.False(VulkanDevice.ResolveSynchronousPipelines(null, null)); + Assert.False(VulkanDevice.ResolveSynchronousPipelines(null, "0")); + Assert.True(VulkanDevice.ResolveSynchronousPipelines(null, "1")); + Assert.True(VulkanDevice.ResolveSynchronousPipelines(null, " true ")); + Assert.True(VulkanDevice.ResolveSynchronousPipelines(true, "0")); + Assert.False(VulkanDevice.ResolveSynchronousPipelines(false, "1")); + } + + /// + /// A parity frame written to disk must hold every draw. The setting works + /// without a capture script and an explicit pipeline policy still wins. + /// + [Fact] + public void AFrameCaptureForcesBlockingPipelinesWithoutTheScripts() + { + Assert.True(VulkanDevice.ResolveSynchronousPipelines(null, null, parityDump: "/tmp/dump")); + Assert.True(VulkanDevice.ResolveSynchronousPipelines(null, "0", parityDump: "/tmp/dump")); + // The capture code ignores a relative or blank directory, so the pipelines do too. + Assert.False(VulkanDevice.ResolveSynchronousPipelines(null, null, parityDump: "relative")); + Assert.False(VulkanDevice.ResolveSynchronousPipelines(false, null, parityDump: "/tmp/dump")); + } + + + private VulkanDevice Open(bool synchronous, string? root = null) => GpuTest.CreateDevice(output, device => { + device.SynchronousPipelines = synchronous; device.ShaderCacheDirectory = root; + }); + + // Vary shader constants to avoid routinely reusing an earlier run's implicit driver cache. + private static (string Fragment, byte[] Color) FreshShader() + { + var id = Guid.NewGuid().ToByteArray(); + byte[] color = { (byte)(1 + id[0] % 254), (byte)(1 + id[1] % 254), (byte)(1 + id[2] % 254), 255 }; + string fragment = string.Format(CultureInfo.InvariantCulture, """ + #version 330 core + out vec4 color; + void main() {{ color = vec4({0}.0 / 255.0, {1}.0 / 255.0, {2}.0 / 255.0, 1); }} + """, color[0], color[1], color[2]); + return (fragment, color); + } + + private static (int Image, int Framebuffer) Target(VulkanDevice device) + { + int image = device.CreateTexture2DRaw(8, 8, 0x8058, IntPtr.Zero, 4); + int target = device.CreateFramebuffer(8, 8); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + return (image, target); + } + + private static void Begin(VulkanDevice device, int target, int program) + { + device.BeginFrame(); device.BindFramebuffer(target); device.ClearColor(0, 0, 0, 0, 1); + device.SetViewport(0, 0, 8, 8); device.SetDepthTest(false); device.SetDepthMask(false); + device.SetCullFace(false); device.SetBlend(false, EnumBlendMode.Standard); device.UseProgram(program); + } + + private static void Pixels(VulkanDevice device, int image, byte[] color) + { + byte[] pixels = device.ReadBackLevel0ForTests(image); + Assert.Equal(8 * 8 * 4, pixels.Length); + for (int i = 0; i < pixels.Length; i++) Assert.Equal(color[i % 4], pixels[i]); + } + + private static void Drain(GraphicsPipelineCache cache) => + Assert.True(cache.WaitForBackgroundCompiles(TimeSpan.FromSeconds(60)), "Pipeline workers did not finish."); + + [SkippableFact] + public void ColdRequestsCompileOnceAndPublishAtTheNextFrame() + { + var device = Open(false); + try + { + var cache = device.PipelinesForTests; + Skip.IfNot(cache.AsyncCompiles, "Device lacks pipelineCreationCacheControl."); + var shader = FreshShader(); int program = GpuTest.LinkProgram(device, Vertex, shader.Fragment); + var target = Target(device); + Begin(device, target.Framebuffer, program); + device.DrawFullscreenTriangle(); device.DrawFullscreenTriangle(); + Assert.Equal(2, cache.DrawsSkipped); Assert.Equal(1, cache.QueuedCompiles); + Assert.Equal(0, cache.Count); Assert.Equal(0, cache.CompiledSync); + device.Present(); Pixels(device, target.Image, new byte[] { 0, 0, 0, 255 }); + Drain(cache); Assert.Equal(1, cache.CompiledAsync); Assert.Equal(0, cache.Count); + Begin(device, target.Framebuffer, program); Assert.Equal(1, cache.Count); + device.DrawFullscreenTriangle(); device.DrawFullscreenTriangle(); + device.Present(); Pixels(device, target.Image, shader.Color); + Assert.Equal(2, cache.Hits); Assert.Equal(2, cache.DrawsSkipped); + Assert.Equal(1, cache.QueuedCompiles); Assert.Equal(0, cache.PendingCompiles); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableTheory] + [InlineData("completed")] + [InlineData("queued")] + [InlineData("deleted")] + public void PersistedProgramsPrewarmWithoutLosingDrawsOrPublishingDeletedPrograms(string state) + { + string root = Path.Combine(Path.GetTempPath(), "optimum-prewarm-" + Guid.NewGuid().ToString("N")); + var shader = FreshShader(); + try + { + var first = Open(true, root); + PipelineCacheIdentity identity; + try + { + Skip.IfNot(first.ContextForTests.Capabilities.PipelineCreationCacheControl, "Device lacks pipelineCreationCacheControl."); + identity = PipelineCacheIdentity.Of(first.ContextForTests.Capabilities); + int program = GpuTest.LinkProgram(first, Vertex, shader.Fragment); + var target = Target(first); Begin(first, target.Framebuffer, program); + first.DrawFullscreenTriangle(); first.Present(); Pixels(first, target.Image, shader.Color); + Assert.Equal(1, first.PipelinesForTests.CompiledSync); + } + finally { first.Dispose(); } + GpuTest.AssertClean(first); + Assert.Equal(1, PipelineKeyLog.Load(PipelineKeyLog.PathFor(root, identity)).Count); + var second = Open(false, root); + try + { + var cache = second.PipelinesForTests; + Assert.True(cache.AsyncCompiles); Assert.True(cache.SeedAccepted); + cache.HoldBackgroundCompilesForTests = state == "queued"; + int program = GpuTest.LinkProgram(second, Vertex, shader.Fragment); + Assert.Equal(1, cache.PendingCompiles); + if (state != "queued") Drain(cache); + if (state == "deleted") second.DeleteProgram(program); + var target = Target(second); + Begin(second, target.Framebuffer, state == "deleted" ? 0 : program); + if (state != "deleted") second.DrawFullscreenTriangle(); + second.Present(); + Pixels(second, target.Image, state == "deleted" ? new byte[] { 0, 0, 0, 255 } : shader.Color); + Assert.Equal(0, cache.DrawsSkipped); Assert.Equal(0, cache.CompiledSync); + Assert.Equal(0, cache.CompiledAsync); Assert.Equal(0, cache.QueuedCompiles); + if (state == "completed") Assert.Equal(1, cache.PrewarmHits); + if (state == "queued") Assert.Equal(1, cache.Warm); + cache.HoldBackgroundCompilesForTests = false; Drain(cache); + second.BeginFrame(); second.Present(); + Assert.Equal(0, cache.PendingCompiles); Assert.Equal(0, cache.PrewarmedWaiting); + Assert.Equal(state == "deleted" ? 0 : 1, cache.Count); + if (state == "queued") Assert.Equal(0, cache.Prewarmed); + } + finally { second.PipelinesForTests.HoldBackgroundCompilesForTests = false; second.Dispose(); } + GpuTest.AssertClean(second); + } + finally { DeleteRoot(root); } + } + + private static void DeleteRoot(string root) + { + if (Directory.Exists(root)) Directory.Delete(root, recursive: true); + } +} diff --git a/Optimum.Render.Vulkan.Tests/Platform/ModPassTests.cs b/Optimum.Render.Vulkan.Tests/Platform/ModPassTests.cs new file mode 100644 index 00000000..dba70625 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Platform/ModPassTests.cs @@ -0,0 +1,652 @@ +// Source: Optimum.Render.Vulkan.Tests/ModPassHostingTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using System.Runtime.CompilerServices; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Graph; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Tests.Fixtures; +using Optimum.Render.Vulkan.Tests.Fixtures.ModPassFixture; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.API.Config; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +/// The mod pass registry and the motion hooks are process statics: these tests run alone. +[CollectionDefinition(Name, DisableParallelization = true)] +public sealed class ModPassCollection +{ + public const string Name = "Optimum mod passes (process statics)"; +} + +/// +/// Vulkan-native plan, Phase 5: mod-declared passes and motion writers. The fixture mod +/// (Fixtures/ModPassFixture) registers one AfterOIT pass that samples Primary's glow, writes +/// Primary's colour and depth and is a motion writer, plus one renderer motion writer. On Vulkan +/// the platform runs the pass at the end of the AfterOIT bracket and nowhere else, with glow out of +/// the scope and shader-readable, colour, depth and motion attached, and the motion window open +/// around the draw; the result reads back after a multi-frame run with zero validation messages. +/// On OpenGL nothing of it runs. +/// +[Collection(ModPassCollection.Name)] +public class ModPassHostingTests +{ + private readonly ITestOutputHelper _output; + + public ModPassHostingTests(ITestOutputHelper output) => _output = output; + + private const int Size = 8; + private static readonly float[] GlowClear = { 0.8f, 0.4f, 0.2f, 1f }; + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + private const string TintFragment = """ + #version 330 core + uniform sampler2D glowTex; + layout(location = 0) out vec4 outColor; + layout(location = 2) out vec4 outMotion; + void main(void) + { + vec4 glow = texelFetch(glowTex, ivec2(gl_FragCoord.xy), 0); + outColor = vec4(glow.rgb * 0.5, 1.0); + outMotion = vec4(1.5, -2.5, 0.25, gl_FragCoord.z); + } + """; + + private static ClientMain HeadlessGame(ClientPlatformAbstract platform) + { + ScreenManager.FrameProfiler ??= new FrameProfilerUtil(static (string _) => { }); + var game = (ClientMain)RuntimeHelpers.GetUninitializedObject(typeof(ClientMain)); + game.Platform = platform; + return game; + } + + private sealed class DrawRecord + { + public EnumRenderStage Stage; + public bool InStage; + public bool MotionWindow; + public string? PassName; + public ImageLayout Colour, Glow, Motion, Depth; + public long UndeclaredSplits; + } + + [SkippableFact] + public void TheFixturePassRunsAtItsSlotWithTheDeclaredAttachmentStatesAndNoValidationMessages() + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-mod-pass-test-" + Guid.NewGuid().ToString("N")); + var platform = new VulkanClientPlatform(null!) + { + DeviceFactory = GpuTest.NewDevice, + CrashMarkerDataPath = dataPath, + }; + (ICoreClientAPI api, ClientApiStub stub) = ClientApiStub.Create(); + var fixture = new ModPassFixtureSystem(); + bool taa = OptimumConfig.Taa; + try + { + bool installed = platform.InitializeGraphics(IntPtr.Zero, 0, 0, out string reason); + if (!installed) _output.WriteLine("Vulkan unavailable: " + reason); + Skip.IfNot(installed, "No usable Vulkan device."); + VulkanDevice seam = platform.GraphicsDevice!; + Assert.NotNull(OptimumModPasses.MotionBeginHook); + + OptimumConfig.Taa = true; + Assert.True(OptimumConfig.EffectiveTaa, "TAA is explicitly disabled by a launcher scan on this machine"); + FrameBufferRef primary = CreatePrimary(seam); + InstallFrameBuffers(platform, primary); + + int program = GpuTest.LinkProgram(seam, FullscreenVertex, TintFragment, "mod-pass-fixture"); + seam.SetSamplerUnit(program, "glowTex", 0); + + fixture.StartClientSide(api); + Assert.Single(OptimumModPasses.ForSlot(EnumOptimumPass.AfterOIT)); + Assert.Single(stub.LeaveWorldHandlers); + + var draws = new List(); + FrameGraph graph = seam.FrameGraphForTests; + fixture.Drawer = _ => + { + seam.SetViewport(0, 0, Size, Size); + platform.GlEnableDepthTest(); + platform.GlDepthMask(true); + platform.GlDisableCullFace(); + platform.GlToggleBlend(false); + platform.UseShaderProgram(program); + seam.BindTexture(0, primary.ColorTextureIds[1]); + seam.DrawFullscreenTriangle(); + draws.Add(new DrawRecord + { + Stage = platform.CurrentRenderStage, + InStage = platform.InRenderStage, + MotionWindow = platform.OptimumMotionWriteActive, + PassName = platform.CurrentModPass, + Colour = seam.TextureLayoutForTests(primary.ColorTextureIds[0]), + Glow = seam.TextureLayoutForTests(primary.ColorTextureIds[1]), + Motion = seam.TextureLayoutForTests(primary.ColorTextureIds[2]), + Depth = seam.TextureLayoutForTests(primary.DepthTextureId), + UndeclaredSplits = graph.UndeclaredSplits, + }); + }; + + ClientMain game = HeadlessGame(platform); + const int frames = 4; + long declaredBefore = graph.DeclaredPasses; + OptimumTemporal.Frame.JitterActive = true; + for (int frame = 0; frame < frames; frame++) + { + platform.BeginFrame(); + platform.CurrentFrameBuffer = primary; + seam.SetDrawBuffers(primary.FboId, 0b111); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearColor(1, GlowClear[0], GlowClear[1], GlowClear[2], GlowClear[3]); + seam.ClearColor(2, 0f, 0f, 0f, 0f); + seam.ClearDepth(1f); + seam.SetDrawBuffers(primary.FboId, 0b011); + + game.TriggerRenderStage(EnumRenderStage.Before, 0.016f); + game.TriggerRenderStage(EnumRenderStage.Opaque, 0.016f); + game.TriggerRenderStage(EnumRenderStage.OIT, 0.016f); + game.TriggerRenderStage(EnumRenderStage.AfterOIT, 0.016f); + Assert.Same(primary, platform.CurrentFrameBuffer); + Assert.False(platform.OptimumMotionWriteActive, "the window closes with the pass"); + + // A renderer's registered writer opens the window inside Opaque on Primary only. + platform.BeginRenderStage(EnumRenderStage.Opaque); + platform.CurrentFrameBuffer = primary; + Assert.False(OptimumModPasses.BeginMotionWriter(new OptimumMotionWriterDecl { Name = "unregistered" })); + Assert.True(OptimumModPasses.BeginMotionWriter(fixture.Writer!)); + Assert.True(platform.OptimumMotionWriteActive); + OptimumModPasses.EndMotionWriter(); + Assert.False(platform.OptimumMotionWriteActive); + platform.EndRenderStage(EnumRenderStage.Opaque); + + OptimumTemporal.Frame.JitterActive = false; + game.TriggerRenderStage(EnumRenderStage.AfterPostProcessing, 0.016f); + platform.BeginRenderStage(EnumRenderStage.AfterFinalComposition); + platform.CurrentFrameBuffer = primary; + Assert.False(OptimumModPasses.BeginMotionWriter(fixture.Writer!), "no motion window outside the temporal window"); + platform.EndRenderStage(EnumRenderStage.AfterFinalComposition); + game.TriggerRenderStage(EnumRenderStage.Ortho, 0.016f); + OptimumTemporal.Frame.JitterActive = true; + platform.EndFrame(); + } + OptimumTemporal.Frame.JitterActive = false; + + Assert.Equal(frames, draws.Count); + foreach (DrawRecord draw in draws) + { + Assert.Equal(EnumRenderStage.AfterOIT, draw.Stage); + Assert.True(draw.InStage); + Assert.True(draw.MotionWindow, "the pass is a motion writer"); + Assert.Equal("Mod/" + fixture.ModId + "/" + ModPassFixtureSystem.PassName + "/0", draw.PassName); + Assert.Equal(ImageLayout.ColorAttachmentOptimal, draw.Colour); + Assert.Equal(ImageLayout.ShaderReadOnlyOptimal, draw.Glow); + Assert.Equal(ImageLayout.ColorAttachmentOptimal, draw.Motion); + Assert.Equal(ImageLayout.DepthAttachmentOptimal, draw.Depth); + Assert.Equal(0, draw.UndeclaredSplits); + } + Assert.Equal(frames, platform.ModPassesRun); + Assert.Equal(0, platform.ModPassesSkipped); + Assert.True(graph.DeclaredPasses > declaredBefore); + Assert.Equal(0, graph.UndeclaredSplits); + + seam.BeginFrame(); + byte[] colour = seam.ReadBackLevel0ForTests(primary.ColorTextureIds[0]); + byte[] glow = seam.ReadBackLevel0ForTests(primary.ColorTextureIds[1]); + byte[] motion = seam.ReadBackLevel0ForTests(primary.ColorTextureIds[2]); + byte[] depth = seam.ReadBackLevel0ForTests(primary.DepthTextureId); + seam.Present(); + + for (int p = 0; p < Size * Size; p++) + { + for (int c = 0; c < 3; c++) + { + double glowByte = Math.Round(GlowClear[c] * 255); + Assert.InRange((double)glow[p * 4 + c], glowByte - 0.5, glowByte + 0.5); + Assert.InRange((double)colour[p * 4 + c], glowByte * 0.5 - 1.1, glowByte * 0.5 + 1.1); + } + Assert.Equal(255, colour[p * 4 + 3]); + Assert.Equal(1.5f, (float)BitConverter.ToHalf(motion, p * 8)); + Assert.Equal(-2.5f, (float)BitConverter.ToHalf(motion, p * 8 + 2)); + Assert.Equal(0.25f, (float)BitConverter.ToHalf(motion, p * 8 + 4)); + } + float[] depths = MemoryMarshal.Cast(depth).ToArray(); + for (int p = 0; p < Size * Size; p++) + { + Assert.Equal(0.5f, depths[p]); + Assert.InRange((float)BitConverter.ToHalf(motion, p * 8 + 6), 0.4995f, 0.5005f); + } + + // Zero validation errors and zero synchronization messages (sync,best on). What remains + // are the device-wide best-practices advisories every GPU test's device reports + // (vendor memory-priority, D32 format, push-constant range); anything else fails. + GpuTest.AssertClean(seam); + foreach (string message in ValidationAssert.Snapshot(GpuTest.MessagesOf(seam))) + { + Assert.True(message.StartsWith("[warning] [BestPractices-", StringComparison.Ordinal), "validation message: " + message); + Assert.DoesNotContain("SYNC-", message); + } + + // Leaving the world unloads the mod: its registrations go with it. + stub.LeaveWorld(); + Assert.Empty(OptimumModPasses.ForSlot(EnumOptimumPass.AfterOIT)); + Assert.Empty(stub.LeaveWorldHandlers); + } + finally + { + OptimumTemporal.Frame.JitterActive = false; + OptimumConfig.Taa = taa; + fixture.Dispose(); + platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + Assert.Null(OptimumModPasses.MotionBeginHook); + } + + [Fact] + public void TheOpenGlPlatformIgnoresRegistrations() + { + (ICoreClientAPI api, ClientApiStub _) = ClientApiStub.Create(); + var fixture = new ModPassFixtureSystem(); + int draws = 0; + fixture.Drawer = _ => draws++; + try + { + fixture.StartClientSide(api); + Assert.Single(OptimumModPasses.ForSlot(EnumOptimumPass.AfterOIT)); + + var platform = new ClientPlatformWindows(null!); + ClientMain game = HeadlessGame(platform); + foreach (EnumRenderStage stage in (EnumRenderStage[])Enum.GetValues(typeof(EnumRenderStage))) + game.TriggerRenderStage(stage, 0.016f); + + Assert.Equal(0, draws); + Assert.Null(OptimumModPasses.MotionBeginHook); + Assert.False(OptimumModPasses.BeginMotionWriter(fixture.Writer!)); + OptimumModPasses.EndMotionWriter(); + Assert.False(platform.OptimumMotionWriteActive); + + // No member of the GL platform names the registry. + foreach (MethodInfo method in typeof(ClientPlatformWindows).GetMethods( + BindingFlags.Public | BindingFlags.NonPublic | BindingFlags.Instance | BindingFlags.Static | BindingFlags.DeclaredOnly)) + { + foreach (ParameterInfo parameter in method.GetParameters()) + Assert.NotEqual(typeof(OptimumPassDecl), parameter.ParameterType); + } + } + finally + { + fixture.Dispose(); + } + Assert.Empty(OptimumModPasses.ForSlot(EnumOptimumPass.AfterOIT)); + } + + [Fact] + public void AHeadlessVulkanPlatformWithoutADeviceRunsNoModPass() + { + (ICoreClientAPI api, ClientApiStub _) = ClientApiStub.Create(); + var fixture = new ModPassFixtureSystem(); + int draws = 0; + fixture.Drawer = _ => draws++; + try + { + fixture.StartClientSide(api); + var platform = new VulkanClientPlatform(null!); + ClientMain game = HeadlessGame(platform); + game.TriggerRenderStage(EnumRenderStage.AfterOIT, 0.016f); + Assert.Equal(0, draws); + Assert.Equal(0, platform.ModPassesRun); + } + finally + { + fixture.Dispose(); + } + } + + private static FrameBufferRef CreatePrimary(VulkanDevice seam) + { + var primary = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + DepthTextureId = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false), + }; + for (int slot = 0; slot < primary.ColorTextureIds.Length; slot++) + seam.AttachTexture(primary.FboId, (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + slot), + primary.ColorTextureIds[slot], 0); + seam.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + seam.SetDrawBuffers(primary.FboId, 0b011); + Assert.True(seam.CheckFramebufferComplete(primary.FboId, out string status), status); + return primary; + } + + /// Primary at slot 0 with its motion attachment at 2 (no SSAO G-buffer), TAA targets ready. + private static void InstallFrameBuffers(VulkanClientPlatform platform, FrameBufferRef primary) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = primary; + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + platform.SetOptimumMotionAttachmentIndex(2); + typeof(ClientPlatformWindows).GetField("optimumTaaTargetsReady", flags)!.SetValue(platform, true); + Assert.True(platform.TaaTargetsReady); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/PassExclusionTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; +using static Optimum.Render.Vulkan.Tests.GpuTest; + +/// +/// Phase 2 review (2026-09-11): an attachment-subset pass (the final composition leaves +/// Primary 1 out of its scope through ) must treat +/// the left-out slot as what it is, a texture outside the scope: +/// +/// sampling it with its draw buffer off does not close the pass's open scope (it was +/// never in it, so there is nothing to exclude and no split); +/// sampling it with its draw buffer on is not attachment feedback, so it takes no +/// ReadSelf copy and no split; +/// a clear on it with its draw buffer on is not dropped: GL clears the texture, so +/// the clear is promoted and lands before the next use. +/// +/// The same frame with the frame graph off (the slot then stays in the scope) gives the +/// same pixels. +/// +public class PassExclusionTests +{ + private readonly ITestOutputHelper _output; + + public PassExclusionTests(ITestOutputHelper output) => _output = output; + + private const int Size = 8; + + private const string FullscreenVertex = """ + #version 330 core + out vec2 uv; + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.5, 1.0); + uv = vec2((x + 1.0) * 0.5, (y + 1.0) * 0.5); + } + """; + + public static TheoryData FrameGraph => new() { true, false }; + + [SkippableTheory] + [MemberData(nameof(FrameGraph))] + public void ALeftOutSlotIsSampledAndClearedLikeATextureOutsideTheScope(bool frameGraph) + { + VulkanDevice created = NewDevice(); + created.FrameGraphEnabled = frameGraph; + if (!created.Initialize(IntPtr.Zero, 0, 0, out string failureReason)) + { + created.Dispose(); + Skip.If(true, "Vulkan unavailable: " + failureReason); + } + + using VulkanDevice seam = created; + FrameGraph graph = seam.FrameGraphForTests; + int Texture() => seam.CreateTexture2D(Size, Size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int c0 = Texture(), c1 = Texture(); + int target = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, c0, 0); + seam.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment1, c1, 0); + + int constant = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + layout(location = 0) out vec4 outColor; + void main(void) { outColor = vec4(1.0, 0.0, 0.0, 1.0); } + """, "px-constant"); + int copy = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D tex; + in vec2 uv; + layout(location = 0) out vec4 outColor; + void main(void) { outColor = texture(tex, uv); } + """, "px-copy"); + seam.SetSamplerUnit(copy, "tex", 0); + + // Seed: c0 black, c1 grey. + seam.BeginFrame(); + BaseState(seam); + seam.DeclarePass(new PassDeclaration { Name = "Seed", FramebufferId = target }); + seam.SetDrawBuffers(target, 0b11); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.ClearColor(1, 0.2f, 0.2f, 0.2f, 1f); + seam.Present(); + + seam.BeginFrame(); + BaseState(seam); + long scopesBefore = seam.ScopesOpenedForTests; + long splitsBefore = graph.Splits; + long feedbackBefore = seam.FeedbackSplitsForTests; + long copiesBefore = seam.ReadSelfCopiesForTests.Created; + + seam.DeclarePass(new PassDeclaration + { + Name = "Compose", FramebufferId = target, ColorSlots = ~(1u << 1), Reads = new[] { c1 }, + }); + + // Draw buffer of the left-out slot off: the first draw opens the scope, the second samples the slot. + seam.SetDrawBuffers(target, 0b01); + seam.UseProgram(constant); + seam.DrawFullscreenTriangle(); + seam.UseProgram(copy); + seam.BindTexture(0, c1); + seam.DrawFullscreenTriangle(); + long splitsAfterSample = graph.Splits - splitsBefore; + + // Draw buffer on: sampling the left-out slot is still not feedback. + seam.SetDrawBuffers(target, 0b11); + seam.DrawFullscreenTriangle(); + seam.BindTexture(0, 0); + long splitsAfterDrawBufferOn = graph.Splits - splitsBefore; + long copies = seam.ReadSelfCopiesForTests.Created - copiesBefore; + long scopes = seam.ScopesOpenedForTests - scopesBefore; + long feedback = seam.FeedbackSplitsForTests - feedbackBefore; + + // A clear on the left-out slot with its draw buffer on clears it, as GL does. + seam.ClearColor(1, 0f, 0f, 1f, 1f); + seam.EndPass(); + seam.SetDrawBuffers(target, 0b01); + seam.Present(); + + seam.BeginFrame(); + byte[] first = seam.ReadBackLevel0ForTests(c0); + byte[] second = seam.ReadBackLevel0ForTests(c1); + seam.Present(); + + _output.WriteLine($"frameGraph={frameGraph} scopes={scopes} splits_after_sample={splitsAfterSample} " + + $"splits_after_draw_buffer_on={splitsAfterDrawBufferOn} feedback_splits={feedback} readself_copies={copies}"); + + int centre = (Size / 2 * Size + Size / 2) * 4; + Assert.Equal(new byte[] { 51, 51, 51, 255 }, first.AsSpan(centre, 4).ToArray()); + Assert.Equal(new byte[] { 0, 0, 255, 255 }, second.AsSpan(centre, 4).ToArray()); + if (frameGraph) + { + Assert.Equal(0, splitsAfterSample); + Assert.Equal(0, splitsAfterDrawBufferOn); + Assert.Equal(0, feedback); + Assert.Equal(0, copies); + Assert.Equal(1, scopes); + } + AssertClean(seam); + } + + private static void BaseState(VulkanDevice seam) + { + seam.SetViewport(0, 0, Size, Size); + seam.SetScissorEnabled(false); + seam.SetDepthTest(false); + seam.SetDepthMask(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + seam.SetColorMask(true, true, true, true); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/RenderStageHookTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Runtime.CompilerServices; +using Optimum.Render.Vulkan.Graph; +using Optimum.Render.Vulkan.Platform; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; + +/// +/// Vulkan-native plan, Phase 2 (contract C3): the donor ClientMain.TriggerRenderStage brackets +/// each stage with the platform's BeginRenderStage/EndRenderStage, and VulkanClientPlatform +/// records the stage and forwards the bracket to its listener. Headless: the real +/// TriggerRenderStage runs on an uninitialised ClientMain (no event manager, so no renderers) +/// against a platform with no device; neither touches GL or Vulkan. +/// +public class RenderStageHookTests +{ + private sealed class RecordingListener : IRenderStageListener + { + public readonly List Calls = new(); + public VulkanClientPlatform? Platform; + public readonly List Faults = new(); + + public void OnBeginRenderStage(EnumRenderStage stage) + { + Calls.Add("begin " + stage); + if (Platform != null && (!Platform.InRenderStage || Platform.CurrentRenderStage != stage)) + Faults.Add("begin " + stage + " saw stage " + Platform.CurrentRenderStage + " active=" + Platform.InRenderStage); + } + + public void OnEndRenderStage(EnumRenderStage stage) + { + Calls.Add("end " + stage); + if (Platform != null && (Platform.InRenderStage || Platform.CurrentRenderStage != stage)) + Faults.Add("end " + stage + " saw stage " + Platform.CurrentRenderStage + " active=" + Platform.InRenderStage); + } + } + + /// Every stage once, in declaration order (the order MainRenderLoop broadly follows). + private static readonly EnumRenderStage[] FrameStages = (EnumRenderStage[])Enum.GetValues(typeof(EnumRenderStage)); + + private static ClientMain HeadlessGame(ClientPlatformAbstract platform) + { + // TriggerRenderStage marks the profiler first; a disabled one returns immediately. + ScreenManager.FrameProfiler ??= new FrameProfilerUtil(static (string _) => { }); + var game = (ClientMain)RuntimeHelpers.GetUninitializedObject(typeof(ClientMain)); + game.Platform = platform; + return game; + } + + [Fact] + public void AListenerSeesBeginAndEndForEachStageInOrder() + { + var platform = new VulkanClientPlatform(null!); + var listener = new RecordingListener { Platform = platform }; + platform.RenderStageListener = listener; + ClientMain game = HeadlessGame(platform); + + for (int frame = 0; frame < 2; frame++) + { + foreach (EnumRenderStage stage in FrameStages) + { + game.TriggerRenderStage(stage, 0.016f); + Assert.False(platform.InRenderStage); + Assert.Equal(stage, platform.CurrentRenderStage); + } + } + + var expected = new List(); + for (int frame = 0; frame < 2; frame++) + { + foreach (EnumRenderStage stage in FrameStages) + { + expected.Add("begin " + stage); + expected.Add("end " + stage); + } + } + Assert.Equal(expected, listener.Calls); + Assert.Empty(listener.Faults); + Assert.True(FrameStages.Length >= 10); + } + + [Fact] + public void WithoutAListenerTheBracketOnlyTracksTheStage() + { + var platform = new VulkanClientPlatform(null!); + Assert.Null(platform.RenderStageListener); + ClientMain game = HeadlessGame(platform); + + game.TriggerRenderStage(EnumRenderStage.Opaque, 0.016f); + Assert.Equal(EnumRenderStage.Opaque, platform.CurrentRenderStage); + Assert.False(platform.InRenderStage); + + platform.BeginRenderStage(EnumRenderStage.OIT); + Assert.True(platform.InRenderStage); + Assert.Equal(EnumRenderStage.OIT, platform.CurrentRenderStage); + platform.EndRenderStage(EnumRenderStage.OIT); + Assert.False(platform.InRenderStage); + } + + [Fact] + public void TheOpenGlPlatformKeepsTheNeutralBodies() + { + var platform = new ClientPlatformWindows(null!); + ClientMain game = HeadlessGame(platform); + + foreach (EnumRenderStage stage in FrameStages) + game.TriggerRenderStage(stage, 0.016f); + + Assert.Equal(typeof(ClientPlatformAbstract), + typeof(ClientPlatformWindows).GetMethod(nameof(ClientPlatformAbstract.BeginRenderStage))!.DeclaringType); + Assert.Equal(typeof(ClientPlatformAbstract), + typeof(ClientPlatformWindows).GetMethod(nameof(ClientPlatformAbstract.EndRenderStage))!.DeclaringType); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/Platform/RoutingTests.cs b/Optimum.Render.Vulkan.Tests/Platform/RoutingTests.cs new file mode 100644 index 00000000..8531763d --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Platform/RoutingTests.cs @@ -0,0 +1,986 @@ +// Source: Optimum.Render.Vulkan.Tests/PlatformDeviceRoutingTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Platform; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +/// +/// Vulkan-native plan, Phase 1A step 4: ClientPlatformWindows keeps only the GL path, and +/// VulkanClientPlatform overrides every graphics member with device calls. These drive the +/// moved overrides through the platform with no GL context: an override that is missing +/// falls into a GL call and throws, one that reaches the wrong device call changes the pixels. +/// +public class PlatformDeviceRoutingTests +{ + private readonly ITestOutputHelper _output; + + public PlatformDeviceRoutingTests(ITestOutputHelper output) => _output = output; + + /// + /// The runtime self-check (VerifyHost) has to cover every override whose base member + /// only exists in the patched lib - an injected ClientPlatformAbstract virtual or a + /// member virtualized in place on ClientPlatformWindows - or an unpatched lib would + /// bypass it mid-frame instead of failing the install. + /// + [Fact] + public void EveryOverrideOfAPatchedVirtualIsInTheSelfCheck() + { + const BindingFlags flags = BindingFlags.Public | BindingFlags.Instance | BindingFlags.DeclaredOnly; + int checkedOverrides = 0; + foreach (MethodInfo method in typeof(VulkanClientPlatform).GetMethods(flags)) + { + MethodInfo definition = method.GetBaseDefinition(); + if (definition == method || method.IsSpecialName) continue; + Type owner = definition.DeclaringType!; + bool onAbstract = owner == typeof(ClientPlatformAbstract); + if (!(onAbstract && !definition.IsAbstract) && owner != typeof(ClientPlatformWindows)) continue; + + ParameterInfo[] parameters = method.GetParameters(); + bool listed = false; + foreach (VulkanClientPlatform.ExpectedVirtual expected in VulkanClientPlatform.ExpectedVirtuals) + { + if (expected.OnAbstract != onAbstract || expected.Name != method.Name || expected.ParameterTypeNames.Length != parameters.Length) continue; + bool same = true; + for (int i = 0; i < parameters.Length && same; i++) + same = expected.ParameterTypeNames[i] == parameters[i].ParameterType.Name; + listed |= same; + } + Assert.True(listed, owner.Name + "." + method.Name + " is overridden but not in VulkanClientPlatform.ExpectedVirtuals"); + checkedOverrides++; + } + _output.WriteLine("patched virtuals overridden: " + checkedOverrides); + Assert.True(checkedOverrides >= 40); + } + + private abstract class BareAbstract { } + private class BareWindows : BareAbstract { } + private sealed class SealedWindows : BareAbstract { } + + [Fact] + public void PatchedHostIsAcceptedAndMissingOrSealedHostsAreRejected() + { + Assert.True(typeof(VulkanClientPlatform).IsSubclassOf(typeof(ClientPlatformWindows))); + Assert.False(typeof(ClientPlatformWindows).IsSealed); + Assert.True(VulkanClientPlatform.VerifyHost(typeof(ClientPlatformAbstract), typeof(ClientPlatformWindows), out string? reason), reason); + Assert.Null(reason); + Assert.False(VulkanClientPlatform.VerifyHost(typeof(BareAbstract), typeof(BareWindows), out string? missing)); + Assert.Contains("InitializeGraphics", missing); + Assert.False(VulkanClientPlatform.VerifyHost(typeof(BareAbstract), typeof(SealedWindows), out string? sealedReason)); + Assert.Contains("sealed", sealedReason); + Assert.NotNull(new VulkanClientPlatform(null!).Logger); + } + + [Fact] + public void FailedInstallNeverConstructsADevice() + { + var platform = new VulkanClientPlatform(null!); + int created = 0; + platform.DeviceFactory = () => { created++; return GpuTest.NewDevice(); }; + string? previous = Environment.GetEnvironmentVariable(VulkanClientPlatform.ForceInstallFailureVariable); + Environment.SetEnvironmentVariable(VulkanClientPlatform.ForceInstallFailureVariable, "1"); + try + { + Assert.False(platform.InitializeGraphics(IntPtr.Zero, 0, 0, out string reason)); + Assert.Equal("forced by OPTIMUM_VULKAN_FORCE_INSTALL_FAILURE", reason); + Assert.Equal(0, created); Assert.Null(platform.GraphicsDevice); + } + finally { Environment.SetEnvironmentVariable(VulkanClientPlatform.ForceInstallFailureVariable, previous); } + } + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + /// + /// Frame 1: CreateFramebuffer, the CurrentFrameBuffer setter (BindCurrentFrameBuffer), + /// the state setters and ClearFrameBuffer (ClearBoundFrameBuffer) through the platform. + /// Frame 2: a program drawn with RenderFullscreenTriangle under GlScissor/GlScissorFlag + /// over the left half only. Frame 3 reads back: left half the draw colour, right half the + /// clear colour. BeginFrame/EndFrame are the platform's bracket; DisposeFrameBuffer and + /// GLDeleteTexture release everything, and validation (sync, best) stays clean. + /// + [SkippableFact] + public unsafe void FramebufferClearScissorAndDrawReachTheDeviceThroughThePlatform() + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-device-routing-test-" + Guid.NewGuid().ToString("N")); + var platform = new VulkanClientPlatform(null!) + { + DeviceFactory = GpuTest.NewDevice, + CrashMarkerDataPath = dataPath, + }; + try + { + bool installed = platform.InitializeGraphics(IntPtr.Zero, 0, 0, out string reason); + if (!installed) _output.WriteLine("Vulkan unavailable: " + reason); + Skip.IfNot(installed, "No usable Vulkan device."); + VulkanDevice seam = platform.GraphicsDevice!; + const int size = 16; + + var attrs = new FramebufferAttrs("routed", size, size) + { + Attachments = new[] + { + new FramebufferAttrsAttachment + { + AttachmentType = EnumFramebufferAttachment.ColorAttachment0, + Texture = new RawTexture + { + Width = size, + Height = size, + PixelInternalFormat = EnumTextureInternalFormat.Rgba8, + PixelFormat = EnumTexturePixelFormat.Rgba, + MinFilter = EnumTextureFilter.Nearest, + MagFilter = EnumTextureFilter.Nearest, + }, + }, + }, + }; + int program = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + out vec4 outColor; + void main(void) { outColor = vec4(200.0 / 255.0, 40.0 / 255.0, 90.0 / 255.0, 1.0); } + """, "routed-draw"); + + platform.BeginFrame(); + FrameBufferRef target = platform.CreateFramebuffer(attrs); + Assert.True(target.FboId > 0); + Assert.Single(target.ColorTextureIds); + platform.CurrentFrameBuffer = target; + Assert.Same(target, platform.CurrentFrameBuffer); + platform.GlDisableDepthTest(); + platform.GlDisableCullFace(); + platform.GlToggleBlend(false); + platform.ClearFrameBuffer(target, new[] { 20f / 255f, 140f / 255f, 220f / 255f, 1f }, clearDepthBuffer: false); + platform.EndFrame(); + + platform.BeginFrame(); + platform.CurrentFrameBuffer = target; + platform.GlDisableDepthTest(); + platform.GlDisableCullFace(); + platform.GlToggleBlend(false); + platform.UseShaderProgram(program); + platform.GlScissor(0, 0, size / 2, size); + platform.GlScissorFlag(true); + Assert.True(platform.GlScissorFlagEnabled); + platform.RenderFullscreenTriangle(null!); + platform.GlScissorFlag(false); + Assert.False(platform.GlScissorFlagEnabled); + platform.UseShaderProgram(0); + platform.EndFrame(); + + var pixels = new byte[size * size * 4]; + platform.BeginFrame(); + platform.CurrentFrameBuffer = target; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, size, size, (IntPtr)destination); + } + platform.EndFrame(); + + int row = size / 2 * size; + int left = (row + 2) * 4; + int right = (row + size - 3) * 4; + _output.WriteLine($"left RGBA = {pixels[left]}, {pixels[left + 1]}, {pixels[left + 2]}, {pixels[left + 3]}"); + _output.WriteLine($"right RGBA = {pixels[right]}, {pixels[right + 1]}, {pixels[right + 2]}, {pixels[right + 3]}"); + Assert.Equal(new byte[] { 200, 40, 90, 255 }, pixels[left..(left + 4)]); + Assert.Equal(new byte[] { 20, 140, 220, 255 }, pixels[right..(right + 4)]); + + platform.DisposeFrameBuffer(target); + // Linked on the device directly above, so freed there too. + seam.DeleteProgram(program); + GpuTest.AssertClean(seam); + } + finally + { + platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/PlatformLeafRoutingTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.IO; +using Optimum.Render.Vulkan.Platform; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +/// +/// Vulkan-native plan, Phase 1A step 5: the leaf operations the render systems outside the +/// platform used to send through the static device seam are platform virtuals, and the seam +/// is gone. These drive them through VulkanClientPlatform with no GL context - an override +/// that is missing lands in the ClientPlatformWindows GL body and throws - and read back what +/// the device made of them. The fork bridge the forked mods use is exercised the same way. +/// +public class PlatformLeafRoutingTests +{ + private readonly ITestOutputHelper _output; + + public PlatformLeafRoutingTests(ITestOutputHelper output) => _output = output; + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + private sealed class Session : IDisposable + { + private readonly string _dataPath; + + public VulkanClientPlatform Platform { get; } + + public VulkanDevice Seam => Platform.GraphicsDevice!; + + private Session(VulkanClientPlatform platform, string dataPath) + { + Platform = platform; + _dataPath = dataPath; + } + + public static Session? TryOpen(ITestOutputHelper output) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-leaf-routing-test-" + Guid.NewGuid().ToString("N")); + var platform = new VulkanClientPlatform(null!) + { + DeviceFactory = GpuTest.NewDevice, + CrashMarkerDataPath = dataPath, + }; + if (!platform.InitializeGraphics(IntPtr.Zero, 0, 0, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + Assert.True(File.Exists(Path.Combine(dataPath, ".optimum", "vulkan-session.lock"))); + return new Session(platform, dataPath); + } + + public void Dispose() + { + Platform.ShutdownGraphics(); + Assert.Null(Platform.GraphicsDevice); + Assert.False(File.Exists(Path.Combine(_dataPath, ".optimum", "vulkan-session.lock"))); + try + { + Directory.Delete(_dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + } + + /// + /// The backend state and the fork bridge follow the platform's graphics: published by + /// InitializeGraphics, withdrawn by ShutdownGraphics. + /// + [SkippableFact] + public void BackendNameAndForkBridgeFollowTheInstall() + { + Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + try + { + Assert.Equal("Vulkan", session!.Platform.GraphicsBackendName); + Assert.True(OptimumRender.IsVulkan); + Assert.NotNull(OptimumForkGraphics.Active); + } + finally + { + session!.Dispose(); + } + Assert.Null(OptimumForkGraphics.Active); + Assert.False(OptimumRender.IsVulkan); + Assert.Null(session.Platform.GraphicsDevice); + } + + /// + /// LoadTextureFromRgbaPointer uploads a solid colour, ClearTextureRegion blanks its left + /// half, the fork bridge builds and binds a framebuffer over it, and ReadDefaultFramebuffer + /// reads the bound target back: left half zero, right half the uploaded colour. The LOD, + /// sampler-LOD and depth-compare setters and the depth range run on the same texture with + /// no GL context, so a missing override would throw. + /// + [SkippableFact] + public unsafe void TextureUploadRegionClearAndReadbackReachTheDevice() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanClientPlatform platform = session!.Platform; + OptimumForkGraphics fork = OptimumForkGraphics.Active!; + const int size = 16; + + var rgba = new byte[size * size * 4]; + for (int i = 0; i < rgba.Length; i += 4) + { + rgba[i] = 10; + rgba[i + 1] = 200; + rgba[i + 2] = 30; + rgba[i + 3] = 255; + } + int texture; + fixed (byte* pixels = rgba) + { + texture = platform.LoadTextureFromRgbaPointer(size, size, (IntPtr)pixels); + } + Assert.True(texture > 0); + platform.ClearTextureRegion(texture, 0, 0, size / 2, size, new int[size / 2 * size]); + + platform.SetTextureLodBias(new[] { texture }, -0.5f); + platform.SetTextureDepthCompare(texture, 0); + int sampler = session.Seam.CreateSampler(true); + platform.SetSamplerLodBias(sampler, 0.25f); + platform.SetDepthRange(0f, 20000f); + + int framebuffer = fork.CreateFramebuffer(size, size); + fork.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + fork.SetDrawBuffers(framebuffer, 1); + + var read = new byte[size * size * 4]; + platform.BeginFrame(); + fork.BindFramebuffer(framebuffer); + fork.SetViewport(0, 0, size, size); + fixed (byte* destination = read) + { + platform.ReadDefaultFramebuffer(0, 0, size, size, (IntPtr)destination); + } + platform.EndFrame(); + + int row = size / 2 * size; + int left = (row + 2) * 4; + int right = (row + size - 3) * 4; + // The platform's readback is the client's seam, and its contract is the + // OpenGL body's: glReadPixels(..., GL_BGRA, ...). So the R=10, G=200, B=30 + // texel that went in comes back B, G, R, A = 30, 200, 10, 255. Asserting + // RGBA here is what let every Vulkan screenshot and AVI recording ship with + // red and blue exchanged, because Screenshot.GrabScreenshot decodes into an + // SKBitmap declared Bgra8888 (wave-1 review, 2026-09-12). The device-level + // VulkanDevice.ReadDefaultFramebuffer still hands texels back in their + // stored order; the conversion is VulkanClientPlatform's. + _output.WriteLine($"left BGRA = {read[left]}, {read[left + 1]}, {read[left + 2]}, {read[left + 3]}"); + _output.WriteLine($"right BGRA = {read[right]}, {read[right + 1]}, {read[right + 2]}, {read[right + 3]}"); + Assert.Equal(new byte[] { 0, 0, 0, 0 }, read[left..(left + 4)]); + Assert.Equal(new byte[] { 30, 200, 10, 255 }, read[right..(right + 4)]); + + fork.DeleteFramebuffer(framebuffer); + platform.GLDeleteTexture(texture); + session.Seam.DeleteSampler(sampler); + GpuTest.AssertClean(session.Seam); + } + + /// + /// ClearDefaultDepth clears the bound target's depth like glClearBuffer: 0.25 stays 0.25, + /// and ScreenManager's 20000 clamps to 1. + /// + [SkippableFact] + public void ClearDefaultDepthClampsLikeGl() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanClientPlatform platform = session!.Platform; + VulkanDevice seam = session.Seam; + const int size = 8; + + int colour = seam.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = seam.CreateTexture2D(size, size, EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(size, size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, colour, 0); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + seam.SetDrawBuffers(framebuffer, 1); + + float Cleared(float value) + { + platform.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.SetViewport(0, 0, size, size); + seam.SetDepthMask(true); + platform.ClearDefaultDepth(value); + platform.EndFrame(); + // The parity readback only runs inside a frame; the next one sees the clear presented. + platform.BeginFrame(); + OptimumTextureReadback? readback = seam.ReadTextureForParity(depth); + platform.EndFrame(); + Assert.NotNull(readback); + Assert.NotNull(readback!.Floats); + return readback.Floats![size / 2 * size + size / 2]; + } + + float quarter = Cleared(0.25f); + float clamped = Cleared(20000f); + _output.WriteLine("depth after 0.25 = " + quarter + ", after 20000 = " + clamped); + Assert.Equal(0.25f, quarter, 5); + Assert.Equal(1f, clamped, 5); + + seam.DeleteFramebuffer(framebuffer); + seam.DeleteTexture(colour); + seam.DeleteTexture(depth); + GpuTest.AssertClean(seam); + } + + /// + /// The sun probe's query protocol through the platform over several presented frames: + /// GenOcclusionQuery, Begin, a fullscreen draw, End, then TryGetOcclusionQueryResult polled + /// each later frame until it reports - and it reports samples. A second query around no + /// draw reports zero. Present runs between frames and nothing reads back inside the loop. + /// + [SkippableFact] + public void OcclusionQueriesCountSamplesAcrossFrames() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanClientPlatform platform = session!.Platform; + VulkanDevice seam = session.Seam; + const int size = 16; + + int colour = seam.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(size, size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, colour, 0); + seam.SetDrawBuffers(framebuffer, 1); + int program = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + out vec4 outColor; + void main(void) { outColor = vec4(1.0); } + """, "leaf-occlusion"); + + int drawn = platform.GenOcclusionQuery(); + int empty = platform.GenOcclusionQuery(); + Assert.True(drawn > 0 && empty > 0 && drawn != empty); + + platform.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.SetViewport(0, 0, size, size); + platform.GlDisableDepthTest(); + platform.GlDisableCullFace(); + platform.GlToggleBlend(false); + platform.GlColorMask(false, false, false, false); + platform.UseShaderProgram(program); + platform.BeginOcclusionQuery(drawn); + platform.RenderFullscreenTriangle(null!); + platform.EndOcclusionQuery(drawn); + platform.BeginOcclusionQuery(empty); + platform.EndOcclusionQuery(empty); + platform.UseShaderProgram(0); + platform.GlColorMask(true, true, true, true); + platform.EndFrame(); + + bool drawnReported = false; + bool emptyReported = false; + int drawnSamples = 0; + int emptySamples = -1; + for (int frame = 0; frame < 60 && !(drawnReported && emptyReported); frame++) + { + platform.BeginFrame(); + if (!drawnReported) drawnReported = platform.TryGetOcclusionQueryResult(drawn, out drawnSamples); + if (!emptyReported) emptyReported = platform.TryGetOcclusionQueryResult(empty, out emptySamples); + platform.EndFrame(); + } + + _output.WriteLine("drawn: " + drawnReported + " " + drawnSamples + ", empty: " + emptyReported + " " + emptySamples); + Assert.True(drawnReported, "the drawn query never reported within 60 frames"); + Assert.True(emptyReported, "the empty query never reported within 60 frames"); + Assert.True(drawnSamples > 0); + Assert.Equal(0, emptySamples); + + platform.DeleteOcclusionQuery(drawn); + platform.DeleteOcclusionQuery(empty); + seam.DeleteProgram(program); + seam.DeleteFramebuffer(framebuffer); + seam.DeleteTexture(colour); + GpuTest.AssertClean(seam); + } + + /// + /// The OIT targets and pass state through the platform: CreateOitTargets attaches the + /// reveal texture at 0 and the three-layer accumulation array at 3-5 of a three-attachment + /// transparent framebuffer, BeginOitAccumulation enables all six draw buffers and clears + /// 0 and 1 to one and 3-5 to zero, BindOitTextures binds units 6 and 7. The reveal target + /// and the framebuffer's own attachment 1 read back as one; its attachment 2, outside the + /// clear set, keeps its earlier clear colour. + /// + [SkippableFact] + public void OitTargetsAndAccumulationClearsReachTheDevice() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanClientPlatform platform = session!.Platform; + VulkanDevice seam = session.Seam; + const int size = 8; + + RawTexture Colour() => new() + { + Width = size, + Height = size, + PixelInternalFormat = EnumTextureInternalFormat.Rgba8, + PixelFormat = EnumTexturePixelFormat.Rgba, + MinFilter = EnumTextureFilter.Nearest, + MagFilter = EnumTextureFilter.Nearest, + }; + var attrs = new FramebufferAttrs("transparent", size, size) + { + Attachments = new[] + { + new FramebufferAttrsAttachment { AttachmentType = EnumFramebufferAttachment.ColorAttachment0, Texture = Colour() }, + new FramebufferAttrsAttachment { AttachmentType = EnumFramebufferAttachment.ColorAttachment1, Texture = Colour() }, + new FramebufferAttrsAttachment { AttachmentType = EnumFramebufferAttachment.ColorAttachment2, Texture = Colour() }, + }, + }; + int oitProgram = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D OITreveal; + uniform sampler2DArray OITaccumulation; + out vec4 outColor; + void main(void) { outColor = texture(OITreveal, vec2(0.5)) + texture(OITaccumulation, vec3(0.5, 0.5, 0.0)); } + """, "leaf-oit"); + + platform.BeginFrame(); + FrameBufferRef transparent = platform.CreateFramebuffer(attrs); + platform.CurrentFrameBuffer = transparent; + platform.ClearFrameBuffer(transparent, new[] { 40f / 255f, 80f / 255f, 120f / 255f, 1f }, clearDepthBuffer: false); + platform.EndFrame(); + + platform.CreateOitTargets(transparent, 3, out int reveal, out int accum); + Assert.True(reveal > 0 && accum > 0 && reveal != accum); + + platform.BeginFrame(); + platform.CurrentFrameBuffer = transparent; + platform.SetProgramSamplerUnit(oitProgram, "OITaccumulation", 7); + platform.BeginOitAccumulation(transparent); + platform.BindOitTextures(reveal, accum); + platform.EndFrame(); + + byte[] Centre(int textureId) + { + OptimumTextureReadback? readback = seam.ReadTextureForParity(textureId); + Assert.NotNull(readback); + Assert.NotNull(readback!.Bytes); + int offset = (size / 2 * size + size / 2) * 4; + return readback.Bytes![offset..(offset + 4)]; + } + + // The parity readback only runs inside a frame; this one follows the presented clears. + platform.BeginFrame(); + byte[] revealTexel = Centre(reveal); + byte[] second = Centre(transparent.ColorTextureIds[1]); + byte[] third = Centre(transparent.ColorTextureIds[2]); + platform.EndFrame(); + _output.WriteLine("reveal = " + string.Join(",", revealTexel) + "; colour1 = " + string.Join(",", second) + "; colour2 = " + string.Join(",", third)); + Assert.Equal(new byte[] { 255, 255, 255, 255 }, revealTexel); + Assert.Equal(new byte[] { 255, 255, 255, 255 }, second); + Assert.Equal(new byte[] { 40, 80, 120, 255 }, third); + + platform.GLDeleteTexture(accum); + platform.GLDeleteTexture(reveal); + platform.DisposeFrameBuffer(transparent); + seam.DeleteProgram(oitProgram); + GpuTest.AssertClean(seam); + } + + private static MeshData Quad() => new MeshData(4, 6, withNormals: false, withUv: false, withRgba: false, withFlags: false) + { + xyz = new[] { -1f, -1f, 0f, 1f, -1f, 0f, 1f, 1f, 0f, -1f, 1f, 0f }, + VerticesCount = 4, + Indices = new[] { 0, 1, 2, 0, 2, 3 }, + IndicesCount = 6, + }; + + /// + /// Phase 1 review regression: MeshRef.Dispose (which the client and mods call directly as + /// often as DeleteMesh) releases the device mesh, exactly once. Before the fix VAO.Dispose + /// reached an empty DeleteVertexArrayHandles override and every directly disposed mesh + /// leaked its device buffers for the rest of the session. + /// + [SkippableFact] + public void DisposingAMeshRefReleasesTheDeviceMeshOnce() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanClientPlatform platform = session!.Platform; + VulkanDevice seam = session.Seam; + ClientPlatformAbstract previous = ScreenManager.Platform; + ScreenManager.Platform = platform; + try + { + MeshRef direct = platform.UploadMesh(Quad()); + int directId = ((VAO)direct).VaoId; + Assert.NotNull(seam.MeshesForTests.Get(directId)); + direct.Dispose(); + Assert.True(direct.Disposed); + Assert.Null(seam.MeshesForTests.Get(directId)); + + MeshRef viaPlatform = platform.UploadMesh(Quad()); + int viaId = ((VAO)viaPlatform).VaoId; + platform.DeleteMesh(viaPlatform); + Assert.True(viaPlatform.Disposed); + Assert.Null(seam.MeshesForTests.Get(viaId)); + + // The freed ids are reused; disposing the released VAOs again must not free the + // mesh that now holds one of them. + MeshRef survivor = platform.UploadMesh(Quad()); + int survivorId = ((VAO)survivor).VaoId; + direct.Dispose(); + viaPlatform.Dispose(); + platform.DeleteMesh(viaPlatform); + Assert.NotNull(seam.MeshesForTests.Get(survivorId)); + + // The released meshes are destroyed on the timelines like any other resource. + for (int frame = 0; frame < 4; frame++) + { + platform.BeginFrame(); + platform.EndFrame(); + } + Assert.NotNull(seam.MeshesForTests.Get(survivorId)); + survivor.Dispose(); + Assert.Null(seam.MeshesForTests.Get(survivorId)); + } + finally + { + ScreenManager.Platform = previous; + } + GpuTest.AssertClean(seam); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/PlatformProgramRoutingTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.IO; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Platform; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +/// +/// Vulkan-native plan, Phase 1A step 3: ShaderProgramBase and UBO no longer talk to the +/// device or to GL; they call ScreenManager.Platform. These drive the lib's own classes +/// (the donor the Cecil patch transplants) through a VulkanClientPlatform installed as +/// the client's platform, and read the pixels back, so a virtual that stops reaching the +/// device shows up as a wrong colour rather than as a missing call. +/// +public class PlatformProgramRoutingTests +{ + private readonly ITestOutputHelper _output; + + public PlatformProgramRoutingTests(ITestOutputHelper output) => _output = output; + + private sealed class RoutedProgram : ShaderProgramBase + { + public override bool Compile() => true; + } + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + /// The installed platform, the platform it replaced and the crash-marker directory. + private sealed class Session : IDisposable + { + private readonly ClientPlatformAbstract? _previous; + private readonly string _dataPath; + + public VulkanClientPlatform Platform { get; } + public VulkanDevice Seam => Platform.GraphicsDevice!; + + private Session(VulkanClientPlatform platform, ClientPlatformAbstract? previous, string dataPath) + { + Platform = platform; + _previous = previous; + _dataPath = dataPath; + } + + public static Session? TryOpen(ITestOutputHelper output) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-routing-test-" + Guid.NewGuid().ToString("N")); + var platform = new VulkanClientPlatform(null!) + { + DeviceFactory = GpuTest.NewDevice, + CrashMarkerDataPath = dataPath, + }; + if (!platform.InitializeGraphics(IntPtr.Zero, 0, 0, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + TryDelete(dataPath); + return null; + } + + var session = new Session(platform, ScreenManager.Platform, dataPath); + ScreenManager.Platform = platform; + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + ScreenManager.Platform = _previous!; + Platform.ShutdownGraphics(); + TryDelete(_dataPath); + } + + private static void TryDelete(string path) + { + try + { + Directory.Delete(path, true); + } + catch (DirectoryNotFoundException) + { + } + } + } + + private static int ColourTarget(VulkanDevice seam, int size) + { + int texture = seam.CreateTexture2D(size, size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int framebuffer = seam.CreateFramebuffer(size, size); + seam.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + seam.SetDrawBuffers(framebuffer, 0b1); + return framebuffer; + } + + private static void BeginDraw(VulkanDevice seam, int framebuffer, int size) + { + seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + seam.ClearColor(0, 0f, 0f, 0f, 1f); + seam.SetViewport(0, 0, size, size); + seam.SetDepthTest(false); + seam.SetCullFace(false); + seam.SetBlend(false, EnumBlendMode.Standard); + } + + private static unsafe byte[] ReadCentre(VulkanDevice seam, int framebuffer, int size, bool openFrame) + { + var pixels = new byte[size * size * 4]; + if (openFrame) seam.BeginFrame(); + seam.BindFramebuffer(framebuffer); + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, size, size, (IntPtr)destination); + } + if (openFrame) seam.Present(); + int centre = (size / 2 * size + size / 2) * 4; + return new[] { pixels[centre], pixels[centre + 1], pixels[centre + 2], pixels[centre + 3] }; + } + + /// + /// Use, the scalar/vector/integer-vector/matrix setters, Stop and Dispose, each + /// called on the lib's ShaderProgramBase. Every uniform contributes to the colour, + /// so any one of them failing to reach the device changes the pixel. + /// + [SkippableFact] + public void UniformsSetOnAShaderProgramReachTheShaderThroughThePlatform() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanDevice seam = session!.Seam; + const int size = 16; + + var program = new RoutedProgram { PassName = "routed-uniforms" }; + program.ProgramId = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform float red; + uniform vec2 greenBlue; + uniform int alphaOn; + uniform ivec3 offsets; + uniform mat4 transform; + out vec4 outColor; + void main(void) + { + vec4 moved = transform * vec4(0.0, 0.0, 0.0, 1.0); + outColor = vec4(red, + greenBlue.x + float(offsets.y) / 255.0, + greenBlue.y + moved.x, + alphaOn == 1 ? 1.0 : 0.0); + } + """); + foreach (string name in new[] { "red", "greenBlue", "alphaOn", "offsets", "transform" }) + { + int location = seam.GetUniformLocation(program.ProgramId, name); + Assert.True(location >= 0, name + " has no location"); + program.uniformLocations[name] = location; + } + + int framebuffer = ColourTarget(seam, size); + var transform = new float[16]; + transform[0] = transform[5] = transform[10] = transform[15] = 1f; + transform[12] = 30f / 255f; // column-major translation x + + BeginDraw(seam, framebuffer, size); + program.Use(); + Assert.Same(program, ShaderProgramBase.CurrentShaderProgram); + program.Uniform("red", 60f / 255f); + program.Uniform("greenBlue", new Vec2f(100f / 255f, 150f / 255f)); + program.Uniform("alphaOn", 1); + program.Uniform("offsets", new Vec3i(0, 20, 0)); + program.UniformMatrix("transform", transform); + seam.DrawFullscreenTriangle(); + program.Stop(); + Assert.Null(ShaderProgramBase.CurrentShaderProgram); + seam.Present(); + + byte[] pixel = ReadCentre(seam, framebuffer, size, openFrame: false); + _output.WriteLine($"centre RGBA = {pixel[0]}, {pixel[1]}, {pixel[2]}, {pixel[3]}"); + Assert.Equal(new byte[] { 60, 120, 180, 255 }, pixel); + + program.Dispose(); + Assert.True(program.Disposed); + GpuTest.AssertClean(seam); + } + + /// + /// CreateUBO, then the object-range UBO update in three consecutive frames, each bound + /// by ShaderProgramBase.Use and unbound by Stop, with no readback between the frames; + /// then Dispose through the platform. + /// + [SkippableFact] + public void EveryUboUpdatePathReachesTheBlockThroughThePlatform() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanDevice seam = session!.Seam; + const int size = 16; + + var program = new RoutedProgram { PassName = "routed-ubo" }; + program.ProgramId = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + layout(std140) uniform Tint { vec4 tint; }; + out vec4 outColor; + void main(void) { outColor = tint; } + """); + + var ubo = Assert.IsType(session.Platform.CreateUBO(program.ProgramId, 0, "Tint", sizeof(float) * 4)); + Assert.True(ubo.Handle > 0); + Assert.Equal("Tint", ubo.BlockName); + program.ubos["Tint"] = ubo; + + var colours = new[] + { + new byte[] { 25, 75, 125 }, + new byte[] { 210, 15, 45 }, + new byte[] { 90, 160, 230 }, + }; + var framebuffers = new int[colours.Length]; + for (int frame = 0; frame < colours.Length; frame++) + { + framebuffers[frame] = ColourTarget(seam, size); + byte[] c = colours[frame]; + // The object-range overload is the one every caller uses (the entity + // renderers' bone upload). The generic Update overloads are not + // driven here: vanilla pins through GCHandleProvider.Pointer, which is + // GCHandle.ToIntPtr (the handle value, not the data address), so they + // upload garbage on GL and on the device alike, before and after this move. + ubo.Update(new[] { c[0] / 255f, c[1] / 255f, c[2] / 255f, 1f }, 0, sizeof(float) * 4); + + BeginDraw(seam, framebuffers[frame], size); + program.Use(); + seam.DrawFullscreenTriangle(); + program.Stop(); + seam.Present(); + } + + for (int frame = 0; frame < colours.Length; frame++) + { + byte[] pixel = ReadCentre(seam, framebuffers[frame], size, openFrame: true); + Assert.Equal(colours[frame][0], pixel[0]); + Assert.Equal(colours[frame][1], pixel[1]); + Assert.Equal(colours[frame][2], pixel[2]); + } + + ubo.Dispose(); + program.Dispose(); + GpuTest.AssertClean(seam); + } + + /// + /// BindTexture2D with a custom sampler: the program aims the sampler at the unit + /// through the platform (the device never reads uniformLocations for it), Stop clears + /// the sampler override, and Dispose deletes sampler and program. + /// + [SkippableFact] + public unsafe void ATextureBoundOnAShaderProgramIsSampledThroughThePlatform() + { + using Session? session = Session.TryOpen(_output); + Skip.If(session == null, "No usable Vulkan device."); + VulkanDevice seam = session!.Seam; + const int size = 16; + + var program = new RoutedProgram { PassName = "routed-texture" }; + program.ProgramId = GpuTest.LinkProgram(seam, FullscreenVertex, """ + #version 330 core + uniform sampler2D source; + out vec4 outColor; + void main(void) { outColor = texture(source, vec2(0.5, 0.5)); } + """); + program.SetCustomSampler("source", false); + Assert.True(program.customSamplers["source"] > 0); + + var texel = new byte[] { 10, 200, 90, 255 }; + int texture; + fixed (byte* data = texel) + { + texture = seam.CreateTexture2D(1, 1, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)data, false); + } + int framebuffer = ColourTarget(seam, size); + + BeginDraw(seam, framebuffer, size); + program.Use(); + program.BindTexture2D("source", texture, 0); + seam.DrawFullscreenTriangle(); + program.Stop(); + seam.Present(); + + byte[] pixel = ReadCentre(seam, framebuffer, size, openFrame: false); + _output.WriteLine($"centre RGBA = {pixel[0]}, {pixel[1]}, {pixel[2]}, {pixel[3]}"); + Assert.Equal(new byte[] { 10, 200, 90, 255 }, pixel); + + program.Dispose(); + GpuTest.AssertClean(seam); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/PresentationTests.cs b/Optimum.Render.Vulkan.Tests/PresentationTests.cs new file mode 100644 index 00000000..ff267285 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/PresentationTests.cs @@ -0,0 +1,206 @@ +using System; +using System.Collections.Generic; +using System.Diagnostics; +using OpenTK.Windowing.GraphicsLibraryFramework; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class PresentationTests(ITestOutputHelper output) +{ + private sealed class Clock : ITimelineClock + { + public ulong FrameRecorded { get; set; } + public ulong TransferRecorded { get; set; } + public ulong FrameCompleted { get; set; } + public ulong TransferCompleted { get; set; } + } + + private sealed class Resource(Action destroy) : IDisposable + { + public void Dispose() => destroy(); + } + + [Fact] + public void RetirementRequiresBothRenderingAndPresentationCompletion() + { + var clock = new Clock(); + var queue = new SwapchainRetirement(clock); + var disposed = new List(); + bool fence = false; + queue.Retire(new Resource(() => disposed.Add(1)), 7, () => fence); + queue.NoteSuccessorReacquired(8); + clock.FrameCompleted = 100; + Assert.Equal(0, queue.Collect()); // Timeline progress cannot replace the fence. + fence = true; + Assert.Equal(1, queue.Collect()); + queue.Retire(new Resource(() => disposed.Add(2)), 101, () => true); + Assert.Equal(0, queue.Collect()); // Nor can the fence replace render completion. + clock.FrameCompleted = 101; + Assert.Equal(1, queue.Collect()); + Assert.Equal(new[] { 1, 2 }, disposed); + Assert.Equal(0, queue.Collect()); + } + + [Fact] + public void ResizeStormsWithoutFencesWaitForASuccessorReacquisition() + { + var clock = new Clock { FrameCompleted = 100 }; + var queue = new SwapchainRetirement(clock); + var disposed = new List(); + queue.Retire(new Resource(() => disposed.Add(1)), 7); + queue.Retire(new Resource(() => disposed.Add(2)), 9); + Assert.Equal(0, queue.Collect()); + queue.NoteSuccessorReacquired(101); + queue.Retire(new Resource(() => disposed.Add(3)), 102); + Assert.Equal(0, queue.Collect()); + clock.FrameCompleted = 110; + Assert.Equal(2, queue.Collect()); + Assert.Equal(new[] { 1, 2 }, disposed); + queue.NoteSuccessorReacquired(111); + clock.FrameCompleted = 111; + Assert.Equal(1, queue.Collect()); + queue.Retire(new Resource(() => disposed.Add(4)), 0); + Assert.Equal(1, queue.Collect()); + Assert.Equal(new[] { 1, 2, 3, 4 }, disposed); + } + + [Fact] + public void AcquireSemaphoresCannotBeReissuedWhileTheirWaitIsPending() + { + var pool = new AcquireSemaphoreFreeList(new ulong[] { 11, 22 }); + ulong first = pool.Take(0), second = pool.Take(0); + pool.ReturnAfter(first, 8); + pool.Return(second); // Failed acquisition did not signal it. + Assert.Equal(second, pool.Take(7)); + Assert.Throws(() => pool.Take(7)); + Assert.Equal(first, pool.Take(8)); + Assert.Throws(() => pool.Take(8)); + } + + [Fact] + public void SurfaceDecisionsBoundRetriesAndRespectSupportedModes() + { + Assert.Equal(AcquireAction.PresentThenRebuild, SwapchainPolicy.OnAcquire(Result.SuboptimalKhr, 0)); + Assert.Equal(AcquireAction.RebuildAndRetry, SwapchainPolicy.OnAcquire(Result.ErrorOutOfDateKhr, 0)); + Assert.Equal(AcquireAction.SkipFrame, SwapchainPolicy.OnAcquire(Result.ErrorOutOfDateKhr, 1)); + Assert.Equal(AcquireAction.Fail, SwapchainPolicy.OnAcquire(Result.ErrorDeviceLost, 0)); + Assert.True(SwapchainPolicy.IsParked(new Extent2D(0, 20))); + Assert.True(SwapchainPolicy.IsParked(new Extent2D(20, 0))); + var supported = new[] { PresentModeKHR.FifoKhr, PresentModeKHR.MailboxKhr, PresentModeKHR.FifoRelaxedKhr }; + Assert.Equal(PresentModeKHR.FifoKhr, SwapchainPolicy.ChoosePresentMode(true, false, supported)); + Assert.Equal(PresentModeKHR.FifoRelaxedKhr, SwapchainPolicy.ChoosePresentMode(true, true, supported)); + Assert.Equal(PresentModeKHR.MailboxKhr, SwapchainPolicy.ChoosePresentMode(false, false, supported)); + Assert.Equal(PresentModeKHR.FifoKhr, SwapchainPolicy.ChoosePresentMode(false, false, new[] { PresentModeKHR.FifoKhr })); + Assert.Equal(2u, SwapchainPolicy.ChooseImageCount(2, 2, PresentModeKHR.MailboxKhr)); + Assert.Equal(4u, SwapchainPolicy.ChooseImageCount(3, 0, PresentModeKHR.FifoKhr)); + var detector = new MissedVsyncDetector(); + for (int i = 0; i < 240; i++) Assert.False(detector.NoteInterval(16)); + bool promoted = false; + for (int i = 0; i < 120; i++) promoted |= detector.NoteInterval(i % 3 == 0 ? 32 : 16); + Assert.True(promoted); + } + + [SkippableTheory] + [InlineData(0)] + [InlineData(30)] + public unsafe void RealWindowSurvivesResizeVsyncChangesAndResourceRetirement(int acquireDelayMs) + { + Window* window = null; + VulkanDevice? device = null; + try + { + Skip.IfNot(GLFW.Init(), "GLFW initialization unavailable."); + Skip.IfNot(GLFW.VulkanSupported(), "Window system has no Vulkan support."); + GLFW.WindowHint(WindowHintClientApi.ClientApi, ClientApi.NoApi); + GLFW.WindowHint(WindowHintBool.Visible, false); + window = GLFW.CreateWindow(128, 96, "Optimum presentation acceptance", null, null); + Skip.If(window == null, "Window creation unavailable."); + GLFW.GetFramebufferSize(window, out int initialWidth, out int initialHeight); + device = GpuTest.NewDevice(); + var configure = device.ConfigureContextOptions; + device.ConfigureContextOptions = options => + { + configure?.Invoke(options); + options.ValidationFeatures = "sync,best"; + options.AcquireDelayForTests = TimeSpan.FromMilliseconds(acquireDelayMs); + }; + Assert.True(device.Initialize((IntPtr)window, initialWidth, initialHeight, out string reason), reason); + output.WriteLine(device.RendererString); + var context = device.ContextForTests; + Assert.True(context.ValidationEnabled); + Assert.DoesNotContain("NOT APPLIED", context.ValidationSettingsApplied); + output.WriteLine(context.Capabilities.LatencySummary); + var swapchain = device.SwapchainForTests!; + long idle = VulkanStats.WaitCount(WaitSite.DeviceWaitIdle); + ulong previousPresent = 0, previousTimeline = 0; + var sizes = new[] { (128, 96), (193, 129), (160, 120), (128, 96) }; + for (int step = 0; step < sizes.Length; step++) + { + var (width, height) = sizes[step]; + GLFW.SetWindowSize(window, width, height); + GLFW.PollEvents(); + GLFW.GetFramebufferSize(window, out int pixelWidth, out int pixelHeight); + Assert.True(pixelWidth > 0 && pixelHeight > 0); + device.Resize(pixelWidth, pixelHeight); + device.SetVSync(step % 2 == 0); + for (int frame = 0; frame < 8; frame++) + { + device.BeginFrame(); + int scratch = device.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + device.DeleteTexture(scratch); + device.BindDefaultFramebuffer(); + device.ClearColor(0, (step * 8 + frame) / 255f, 0.2f, 0.4f, 1); + if (frame == 3) + { + byte[] pixel = new byte[4]; + fixed (byte* pointer = pixel) device.ReadDefaultFramebuffer(0, 0, 1, 1, (IntPtr)pointer); + Assert.InRange((int)pixel[0], step * 8 + frame - 1, step * 8 + frame + 1); + Assert.InRange((int)pixel[1], 50, 52); + Assert.InRange((int)pixel[2], 101, 103); + } + device.Present(); + var timing = device.LastPresentTimingsForTests; + Assert.True(timing.Presented); + Assert.True(timing.PresentValue > timing.RenderValue && timing.RenderValue > previousTimeline); + previousTimeline = timing.PresentValue; + Assert.True(timing.AcquireReturned >= timing.FrameSubmitted); + if (acquireDelayMs > 0) + Assert.True((timing.AcquireReturned - timing.FrameSubmitted) * 1000.0 / Stopwatch.Frequency >= acquireDelayMs * 0.9); + Assert.True(swapchain.PresentIds.LastPresentId > previousPresent); + previousPresent = swapchain.PresentIds.LastPresentId; + Assert.Equal(device.LatencyFrameId, swapchain.PresentIds.LastFrameId); + Assert.Null(swapchain.RebuildFailure); + } + Assert.Equal((uint)pixelWidth, swapchain.Extent.Width); + Assert.Equal((uint)pixelHeight, swapchain.Extent.Height); + } + // Complete submitted work, then let the next acquisition collect retired + // chains. A fence-enabled device may finish presentation after rendering. + var drain = Stopwatch.StartNew(); + while (swapchain.RetiredPending > 0 && drain.Elapsed < TimeSpan.FromSeconds(5)) + { + device.BeginFrame(); + device.BindDefaultFramebuffer(); + device.ClearColor(0, 0, 0, 0, 1); + device.Present(); + } + Assert.Equal(0, swapchain.RetiredPending); + Assert.True(swapchain.Creations >= sizes.Length); + Assert.Equal(idle, VulkanStats.WaitCount(WaitSite.DeviceWaitIdle)); + GpuTest.AssertClean(device); + } + finally + { + device?.Dispose(); + if (window != null) GLFW.DestroyWindow(window); + GLFW.Terminate(); + } + if (device != null) GpuTest.AssertClean(device); // Includes teardown diagnostics. + } +} diff --git a/Optimum.Render.Vulkan.Tests/RenderDiagnosticsTests.cs b/Optimum.Render.Vulkan.Tests/RenderDiagnosticsTests.cs new file mode 100644 index 00000000..30ae213c --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/RenderDiagnosticsTests.cs @@ -0,0 +1,184 @@ +using System; +using System.Collections.Generic; +using System.Diagnostics; +using System.Globalization; +using System.IO; +using System.Linq; +using System.Threading.Tasks; +using Optimum.Render.Vulkan.Core; +using Optimum.Tests; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class RenderDiagnosticsTests(ITestOutputHelper output) +{ + [Fact] + public void PacingStatisticsUseOnlyTheLatestValidIntervals() + { + var ring = new FrameIntervalRing(100); + for (int i = 0; i < 5; i++) ring.Add(1000); + for (int i = 1; i <= 100; i++) ring.Add(i); + foreach (double invalid in new[] { double.NaN, -1, double.PositiveInfinity }) ring.Add(invalid); + var sample = ring.Snapshot(); + Assert.Equal(100, sample.Samples); + Assert.Equal(50, sample.P50); Assert.Equal(95, sample.P95); Assert.Equal(99, sample.P99); + Assert.Equal(Math.Sqrt(9999.0 / 12), sample.StdDev, 9); + Assert.Equal(0, sample.Stutters); + for (int i = 0; i < 95; i++) ring.Add(10); + foreach (double interval in new[] { 20, 20.001, 25, 30, 100 }) ring.Add(interval); + Assert.Equal(4, ring.Snapshot().Stutters); + Assert.Equal(10, ring.Snapshot().P50); + Assert.Equal(default, new FrameIntervalRing(4).Snapshot()); + } + + [Fact] + public void PhaseDurationsBelongToThePresentedFrame() + { + var tracker = new LatencyPhaseTracker(); + tracker.Mark(42, LatencyMarker.InputSample, 100); + tracker.Mark(42, LatencyMarker.SimulationStart, 105); + tracker.Mark(42, LatencyMarker.SimulationEnd, 125); + tracker.Mark(42, LatencyMarker.RenderSubmitStart, 130); + tracker.Mark(42, LatencyMarker.RenderSubmitEnd, 160); + tracker.Mark(42, LatencyMarker.PresentStart, 165); + tracker.Mark(42, LatencyMarker.PresentEnd, 172); + + Assert.False(tracker.TryComplete(41, 900, out _)); + Assert.True(tracker.TryComplete(42, 901, out var report)); + Assert.Equal(new LatencyFrameReport(42, 901, 5, 20, 30, 7, 72), report); + Assert.False(tracker.TryComplete(42, 902, out _)); + } + + [Fact] + public void InterruptedFramesCannotLeakTimestampsIntoTheNextReport() + { + var warnings = new List(); + var tracker = new LatencyPhaseTracker(warnings.Add); + tracker.Mark(1, LatencyMarker.SimulationStart, 10); + tracker.Mark(2, LatencyMarker.SimulationStart, 100); + tracker.Mark(3, LatencyMarker.SimulationStart, 200); + tracker.Mark(3, LatencyMarker.SimulationEnd, 230); + tracker.Mark(3, LatencyMarker.PresentEnd, 250); + + Assert.True(tracker.TryComplete(3, 7, out var report)); + Assert.Equal(30UL, report.SimulationUs); + Assert.Equal(50UL, report.TotalUs); + Assert.Equal(0UL, report.RenderSubmitUs); + Assert.Equal(0UL, report.PresentUs); + Assert.Single(warnings); + } + + [Fact] + public void UnconsumedReportsStayBoundedAndDrainInFrameOrder() + { + var recorder = new FrameTimingRecorder(); + for (ulong frame = 1; frame <= 300; frame++) + { + recorder.Marker(frame, LatencyMarker.InputSample); + recorder.Marker(frame, LatencyMarker.PresentEnd); + recorder.OnPresent(frame, frame + 1000); + } + var reports = recorder.TakeReports(); + Assert.Equal(256, reports.Length); + Assert.Equal(Enumerable.Range(45, 256).Select(i => (ulong)i), reports.Select(r => r.FrameId)); + Assert.All(reports, r => Assert.Equal(r.FrameId + 1000, r.PresentId)); + Assert.Empty(recorder.TakeReports()); + } + + [Fact] + public void MissingOrReversedPhasesDoNotInventElapsedTime() + { + var report = LatencyFrameReport.FromCpuTimestamps(9, 12, 0, 100, 90, 0, 120, 130, 130); + Assert.Equal(0UL, report.InputUs); + Assert.Equal(0UL, report.SimulationUs); + Assert.Equal(0UL, report.RenderSubmitUs); + Assert.Equal(0UL, report.PresentUs); + Assert.Equal(30UL, report.TotalUs); + } + + [Fact] + public void CpuPhaseReportsKeepInvariantNumbersAndKnownDurations() + { + var reports = new[] { + new LatencyFrameReport(1, 10, 1000, 2000, 3000, 400, 16000), + new LatencyFrameReport(2, 11, 3000, 4000, 5000, 600, 20000), + }; + var mean = new double[VulkanStats.LatencyIntervalCount]; + var p99 = new double[VulkanStats.LatencyIntervalCount]; + VulkanStats.ReduceReports(reports, mean, p99); + Assert.Equal(new[] { 2.0, 3, 4, 0.5, 18 }, mean); + Assert.Equal(new[] { 3.0, 4, 5, 0.6, 20 }, p99); + CultureInfo previous = CultureInfo.CurrentCulture; + try + { + CultureInfo.CurrentCulture = CultureInfo.GetCultureInfo("de-DE"); + var fields = VulkanStats.FormatLatencyLine(2, mean, p99).Split(' ').Skip(1) + .Select(field => field.Split('=')).ToDictionary(field => field[0], field => field[1]); + Assert.Equal("2", fields["frames"]); + Assert.Equal("18.00", fields["total_mean_ms"]); + Assert.Equal("0.60", fields["present_p99_ms"]); + Assert.All(fields.Values, value => Assert.True(double.TryParse(value, NumberStyles.Float, CultureInfo.InvariantCulture, out _))); + } + finally { CultureInfo.CurrentCulture = previous; } + VulkanStats.ReduceReports(Array.Empty(), mean, p99); + Assert.All(mean, value => Assert.Equal(0, value)); + Assert.All(p99, value => Assert.Equal(0, value)); + } + + [Fact] + public void ErrorsAreDeliveredOnceWithoutAllocatingOnTheEmptyPath() + { + using var device = GpuTest.NewDevice(); // Diagnostics do not require a GPU. + device.AddDiagnosticForTests("warning"); + for (int i = 0; i < 1000; i++) device.GetError(); + long before = GC.GetAllocatedBytesForCurrentThread(); + bool unexpected = false; + for (int i = 0; i < 10000; i++) unexpected |= device.GetError() != null; + long allocated = GC.GetAllocatedBytesForCurrentThread() - before; + Assert.False(unexpected); Assert.Equal(0, allocated); + Parallel.For(0, 256, i => device.AddDiagnosticForTests(VulkanContext.ErrorPrefix + i)); + string[] actual = device.GetError()!.Split('\n'); + Assert.Equal(256, actual.Length); + Assert.Equal(Enumerable.Range(0, 256), actual.Select(line => int.Parse(line[VulkanContext.ErrorPrefix.Length..])).Order()); + Assert.Null(device.GetError()); + device.AddDiagnosticForTests(VulkanContext.ErrorPrefix + "first"); + device.AddDiagnosticForTests("advice"); + device.AddDiagnosticForTests(VulkanContext.ErrorPrefix + "second"); + Assert.Equal(VulkanContext.ErrorPrefix + "first\n" + VulkanContext.ErrorPrefix + "second", device.GetError()); + Assert.Null(device.GetError()); + } + + [Theory] + [InlineData(0, 0)] + [InlineData(2, 1)] + public async Task PacingGateAcceptsActualStatsAndRejectsBlockingUploads(int blockingUploads, int expectedExit) + { + string directory = Directory.CreateTempSubdirectory("optimum-pacing-").FullName; + try + { + string fps = Path.Combine(directory, "fps.log"), stats = Path.Combine(directory, "stats.log"); + File.WriteAllText(fps, "[Optimum] fps window=1.002 frames=120 mean=8.350 min=7.900 max=10.100 p99=9.800 stddev=0.500\n"); + File.WriteAllText(stats, string.Join('\n', + VulkanStats.FormatIntervalLine(1, 120, 0, 812, 0, 0, 0, 0, 0, 0), + VulkanStats.FormatPacingLine(new FramePacingSnapshot(512, 8.3, 9.8, 10.7, 0.6, 0)), + VulkanStats.FormatWaitsLine(new long[VulkanStats.WaitSiteCount], new double[VulkanStats.WaitSiteCount]), + VulkanStats.FormatCountersLine(new CounterSample(blockingUploads, 0, 2640, 240, 0, 168000, 402112, 16777216))) + "\n"); + var start = new ProcessStartInfo(TestToolchain.Bash) { + RedirectStandardOutput = true, RedirectStandardError = true, UseShellExecute = false, + CreateNoWindow = true, WorkingDirectory = ShaderCorpus.RepositoryRoot, + }; + foreach (string argument in new[] { Path.Combine(ShaderCorpus.RepositoryRoot, "scripts", "dev", "pacing-gate.sh"), + "--renderer", "vulkan", "--fps", fps, "--stats", stats }) start.ArgumentList.Add(argument); + using var process = Process.Start(start)!; + var stdout = process.StandardOutput.ReadToEndAsync(); + var stderr = process.StandardError.ReadToEndAsync(); + try { await process.WaitForExitAsync().WaitAsync(TimeSpan.FromSeconds(30)); } + catch { process.Kill(entireProcessTree: true); throw; } + output.WriteLine(await stdout + await stderr); + Assert.Equal(expectedExit, process.ExitCode); + } + finally { Directory.Delete(directory, recursive: true); } + } +} diff --git a/Optimum.Render.Vulkan.Tests/Rendering/MeshTests.cs b/Optimum.Render.Vulkan.Tests/Rendering/MeshTests.cs new file mode 100644 index 00000000..66f9d50f --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Rendering/MeshTests.cs @@ -0,0 +1,857 @@ +// Source: Optimum.Render.Vulkan.Tests/MeshDrawRangeTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Xunit; + +public class MeshDrawRangeTests +{ + [Fact] + public void PoolOffsetsArePointerSizedAndStayPairedWithTheirCounts() + { + // MeshDataPool allocates two ints per GL pointer and writes the low + // word at group * 2. Spare capacity is not another group to draw. + int[] starts = { 48, 0, 28536, 0, 21600, 0, 123456, 0 }; + int[] sizes = { 384, 5400, 144, 999 }; + var commands = new DrawIndexedIndirectCommand[3]; + + MeshManager.WriteIndirectCommands(commands, starts, sizes); + + Assert.Equal(new uint[] { 12, 7134, 5400 }, Array.ConvertAll(commands, c => c.FirstIndex)); + Assert.Equal(new uint[] { 384, 5400, 144 }, Array.ConvertAll(commands, c => c.IndexCount)); + Assert.All(commands, command => + { + Assert.Equal(1u, command.InstanceCount); + Assert.Equal(0, command.VertexOffset); + Assert.Equal(0u, command.FirstInstance); + }); + } + + [Fact] + public void ZeroGroupsNeedNoOffsets() + { + MeshManager.WriteIndirectCommands(Span.Empty, + ReadOnlySpan.Empty, ReadOnlySpan.Empty); + } + + [Fact] + public void LargeByteOffsetsRetainBothWordsBeforeConversionToAnIndex() + { + var commands = new DrawIndexedIndirectCommand[2]; + MeshManager.WriteIndirectCommands(commands, + new[] { unchecked((int)0x80000000), 0, 24, 1 }, new[] { 6, 12 }); + Assert.Equal(0x20000000u, commands[0].FirstIndex); + Assert.Equal(0x40000006u, commands[1].FirstIndex); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/MeshManagerTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Xunit; +using Xunit.Abstractions; + +/// +/// Covers mesh creation and drawing. +/// +/// The layout derivation is the part worth pinning. The game allocates one buffer +/// per attribute and assigns attribute locations by walking the parts in a fixed +/// order, skipping absent ones - so a mesh with positions and colours but no +/// normals or UVs puts colours at location 1. The chunk shaders' explicit +/// locations depend on exactly that, and getting it wrong renders garbage rather +/// than failing. +/// +public class MeshManagerTests +{ + private readonly ITestOutputHelper _output; + + public MeshManagerTests(ITestOutputHelper output) => _output = output; + + private static bool TryCreateContext( + ITestOutputHelper output, List messages, out VulkanContext? context) => + GpuTest.TryCreateContext(output, messages, out context); + + [SkippableTheory] + [InlineData(false)] + [InlineData(true)] + public void PooledTopsoilUsesUnsignedNormalizedShortUvs(bool ssbo) + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + using (context) + { + using var meshes = new MeshManager(context!); + var uv2 = new CustomMeshDataPartShort(8) + { + InterleaveSizes = new[] { 2 }, + InterleaveOffsets = new[] { 0 }, + InterleaveStride = 4, + Conversion = DataConversion.NormalizedFloat, + }; + int mesh = meshes.CreateEmpty(48, 0, 32, 16, 16, 24, + null, uv2, null, null, EnumDrawMode.Triangles, staticDraw: false, ssbo: ssbo); + + VertexLayoutDescription layout = meshes.Get(mesh)!.Layout; + Assert.Equal(Format.R16G16Unorm, layout.Attributes[^1].Format); + Assert.Equal(4u, layout.Bindings[^1].Stride); + } + } + + [SkippableTheory] + [InlineData(DataConversion.NormalizedFloat, false, Format.R16G16Unorm)] + [InlineData(DataConversion.Float, false, Format.R16G16Uscaled)] + [InlineData(DataConversion.Integer, false, Format.R16G16Sint)] + [InlineData(DataConversion.NormalizedFloat, true, Format.R16G16SNorm)] + [InlineData(DataConversion.Float, true, Format.R16G16Sscaled)] + [InlineData(DataConversion.Integer, true, Format.R16G16Sint)] + public void ShortSignednessMatchesTheTwoGlAllocationPaths(DataConversion conversion, bool uploaded, Format expected) + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + using (context) + { + using var meshes = new MeshManager(context!); + var shorts = new CustomMeshDataPartShort(8) + { + InterleaveSizes = new[] { 2 }, + InterleaveOffsets = new[] { 0 }, + InterleaveStride = 4, + Conversion = conversion, + Instanced = true, + }; + int mesh = meshes.CreateEmpty(48, 0, 0, 0, 0, 24, + null, shorts, null, null, EnumDrawMode.Triangles, staticDraw: false, + ssbo: false, signedCustomShorts: uploaded); + VertexLayoutDescription layout = meshes.Get(mesh)!.Layout; + Assert.Equal(expected, layout.Attributes[^1].Format); + Assert.True(layout.Bindings[^1].PerInstance); + } + } + + [SkippableFact] + public void AbsentPartsDoNotConsumeAttributeLocations() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + // Positions and colours only: no normals, no UVs, no flags. + int mesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 0, + rgbaSize: 4 * 4, flagsSize: 0, indicesSize: 6 * sizeof(int), + null, null, null, null, EnumDrawMode.Triangles, staticDraw: true, ssbo: false); + + VertexLayoutDescription layout = meshes.LayoutOf(meshes.LayoutIdOf(mesh)); + + Assert.Equal(2, layout.Bindings.Length); + Assert.Equal(2, layout.Attributes.Length); + + // Colours take location 1 because normals and UVs were absent. + Assert.Equal(0u, layout.Attributes[0].Location); + Assert.Equal(Format.R32G32B32Sfloat, layout.Attributes[0].Format); + Assert.Equal(1u, layout.Attributes[1].Location); + Assert.Equal(Format.R8G8B8A8Unorm, layout.Attributes[1].Format); + } + } + + /// + /// The full chunk-style layout: positions, UVs, colours, flags and a custom + /// integer part, which is what chunkopaque.vsh declares at locations 0 to 4. + /// + [SkippableFact] + public void TheChunkVertexLayoutMatchesTheShadersDeclaredLocations() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + // Count stays 0 until values are added, and AllocationSize reports + // Count - so this is exactly the "declared but not yet filled" case + // that must still claim location 4. + var customInts = new CustomMeshDataPartInt(4) + { + InterleaveSizes = new[] { 1 }, + InterleaveOffsets = new[] { 0 }, + InterleaveStride = 4, + Conversion = DataConversion.Integer, + }; + Assert.Equal(0, customInts.AllocationSize); + + int mesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 4 * 2 * sizeof(float), + rgbaSize: 4 * 4, flagsSize: 4 * sizeof(int), indicesSize: 6 * sizeof(int), + null, null, null, customInts, EnumDrawMode.Triangles, staticDraw: true, ssbo: false); + + VertexLayoutDescription layout = meshes.LayoutOf(meshes.LayoutIdOf(mesh)); + var byLocation = layout.Attributes.ToDictionary(a => a.Location, a => a.Format); + + Assert.Equal(Format.R32G32B32Sfloat, byLocation[0]); // xyz + Assert.Equal(Format.R32G32Sfloat, byLocation[1]); // uv + Assert.Equal(Format.R8G8B8A8Unorm, byLocation[2]); // rgbaLight + // Signed, because every chunk shader declares these as `in int`. + // GL let an unsigned attribute pointer feed a signed input by + // reinterpreting the bits; Vulkan requires the attribute format's + // numeric type to match the shader's exactly, and a mismatch is + // undefined behaviour rather than a reinterpretation. + Assert.Equal(Format.R32Sint, byLocation[3]); // renderFlags + Assert.Equal(Format.R32Sint, byLocation[4]); // colormapData + } + } + + /// + /// With SSBO vertex fetch the chunk shaders read positions from a storage + /// buffer keyed on gl_VertexIndex, so the position buffer must leave the + /// vertex input entirely rather than being bound twice. + /// + /// The same goes for normals, UVs and flags: the face record carries all + /// three, and GL's SSBO allocator creates neither a buffer nor an attribute + /// pointer for them. Binding one anyway pushes rgba off location 0, and the + /// chunk shaders declare rgbaLightIn there - so the block light would be + /// read from the UV stream, tinting terrain by its atlas coordinates. + /// + [SkippableFact] + public void TheSsboPathBindsOnlyTheColoursTheChunkShadersDeclare() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + // The pool passes its configured sizes whichever path it is on, so + // normals, UVs and flags all arrive non-zero here. + int mesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 4 * sizeof(int), + uvSize: 4 * 2 * sizeof(float), rgbaSize: 4 * 4, flagsSize: 4 * sizeof(int), + indicesSize: 6 * sizeof(int), + null, null, null, null, EnumDrawMode.Triangles, staticDraw: true, ssbo: true); + + VulkanMesh created = meshes.Get(mesh)!; + VertexLayoutDescription layout = meshes.LayoutOf(created.LayoutId); + + // The position buffer still exists, and still carries the records. + Assert.NotNull(created.Buffers[MeshManager.BufferXyz]); + Assert.DoesNotContain(MeshManager.BufferXyz, created.BindingOrder); + + // The other three do not exist at all, as in GL. + Assert.Null(created.Buffers[MeshManager.BufferNormals]); + Assert.Null(created.Buffers[MeshManager.BufferUv]); + Assert.Null(created.Buffers[MeshManager.BufferFlags]); + + // Leaving colours alone at location 0. + Assert.Single(layout.Attributes); + Assert.Equal(0u, layout.Attributes[0].Location); + Assert.Equal(Format.R8G8B8A8Unorm, layout.Attributes[0].Format); + } + } + + /// + /// The chunk meshes carry two custom ints per vertex: the colormap data and + /// one more. On the SSBO path the record already holds the colormap data, so + /// GL drops that member, halves the stride and halves the buffer - and the + /// remaining member reads from the offset its predecessor used. + /// + [SkippableFact] + public void TheSsboPathDropsTheCustomIntTheFaceRecordAlreadyCarries() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + CustomMeshDataPartInt TwoPerVertex() => new(8) + { + InterleaveSizes = new[] { 1, 1 }, + InterleaveOffsets = new[] { 0, 4 }, + InterleaveStride = 8, + Conversion = DataConversion.Integer, + }; + + int ssboMesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 0, + rgbaSize: 4 * 4, flagsSize: 0, indicesSize: 6 * sizeof(int), + null, null, null, TwoPerVertex(), EnumDrawMode.Triangles, staticDraw: true, ssbo: true); + + VertexLayoutDescription ssbo = meshes.LayoutOf(meshes.LayoutIdOf(ssboMesh)); + + // Colours at 0, then the one surviving int at 1 - reading offset 0 + // on a four-byte stride, which is where the dropped member sat. + Assert.Equal(2, ssbo.Attributes.Length); + Assert.Equal(Format.R32Sint, ssbo.Attributes[1].Format); + Assert.Equal(0u, ssbo.Attributes[1].Offset); + Assert.Equal(4u, ssbo.Bindings[1].Stride); + + // Off the SSBO path the same part keeps both members and its stride. + int plainMesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 0, + rgbaSize: 4 * 4, flagsSize: 0, indicesSize: 6 * sizeof(int), + null, null, null, TwoPerVertex(), EnumDrawMode.Triangles, staticDraw: true, ssbo: false); + + VertexLayoutDescription plain = meshes.LayoutOf(meshes.LayoutIdOf(plainMesh)); + Assert.Equal(4, plain.Attributes.Length); + Assert.Equal(8u, plain.Bindings[^1].Stride); + } + } + + /// + /// A part that only ever had the colormap int has nothing left once it is + /// dropped, so GL binds it nowhere at all on the SSBO path. + /// + [SkippableFact] + public void TheSsboPathDropsASingleCustomIntPartEntirely() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + var customInts = new CustomMeshDataPartInt(4) + { + InterleaveSizes = new[] { 1 }, + InterleaveOffsets = new[] { 0 }, + InterleaveStride = 4, + Conversion = DataConversion.Integer, + }; + + int mesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 0, + rgbaSize: 4 * 4, flagsSize: 0, indicesSize: 6 * sizeof(int), + null, null, null, customInts, EnumDrawMode.Triangles, staticDraw: true, ssbo: true); + + VulkanMesh created = meshes.Get(mesh)!; + Assert.Null(created.Buffers[MeshManager.BufferCustomInt]); + Assert.Single(meshes.LayoutOf(created.LayoutId).Attributes); + } + } + + [SkippableFact] + public void MeshIdsBehaveLikeGlNamesIncludingReuse() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + int first = meshes.CreateEmpty(48, 0, 0, 16, 0, 24, null, null, null, null, + EnumDrawMode.Triangles, true, false); + int second = meshes.CreateEmpty(48, 0, 0, 16, 0, 24, null, null, null, null, + EnumDrawMode.Triangles, true, false); + + Assert.True(first > 0); + Assert.NotEqual(first, second); + Assert.Null(meshes.Get(0)); + + meshes.Delete(first); + Assert.Null(meshes.Get(first)); + Assert.Equal(first, meshes.CreateEmpty(48, 0, 0, 16, 0, 24, null, null, null, null, + EnumDrawMode.Triangles, true, false)); + } + } + + /// + /// Identical layouts intern to one id so they share a pipeline; a different + /// set of parts must not. + /// + [SkippableFact] + public void IdenticalLayoutsShareAnIdAndDifferentOnesDoNot() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + var state = new PipelineKeyState(); + using var meshes = new MeshManager(context!); + + int a = meshes.CreateEmpty(48, 0, 0, 16, 0, 24, null, null, null, null, + EnumDrawMode.Triangles, true, false); + int b = meshes.CreateEmpty(96, 0, 0, 32, 0, 48, null, null, null, null, + EnumDrawMode.Triangles, true, false); + int c = meshes.CreateEmpty(48, 0, 32, 16, 0, 24, null, null, null, null, + EnumDrawMode.Triangles, true, false); + + // Same parts, different sizes: one layout. + Assert.Equal(meshes.LayoutIdOf(a), meshes.LayoutIdOf(b)); + // UVs added: a different layout. + Assert.NotEqual(meshes.LayoutIdOf(a), meshes.LayoutIdOf(c)); + } + } + + /// + /// The end-to-end mesh check: build a quad, write it through the persistent + /// mapping the way the tesselator does, draw it indexed, and read the pixels. + /// + [SkippableFact] + public unsafe void AnIndexedMeshRendersWithItsVertexColours() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + { + const uint size = 16; + using var commands = new SetupQueue(context!); + using var textures = new TextureManager(context!, commands.Uploads); + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var meshes = new MeshManager(context!); + using var compiler = new ShaderCompiler(); + + int target = textures.Create(size, size, Format.R8G8B8A8Unorm); + int framebuffer = targets.Create(size, size); + targets.Attach(framebuffer, 0, target); + + // A full-target quad: positions plus colours, no normals or UVs, so + // colours land at location 1. + int mesh = meshes.CreateEmpty( + xyzSize: 4 * 3 * sizeof(float), normalsSize: 0, uvSize: 0, + rgbaSize: 4 * 4, flagsSize: 0, indicesSize: 6 * sizeof(int), + null, null, null, null, EnumDrawMode.Triangles, staticDraw: false, ssbo: false); + + float[] positions = + { + -1f, -1f, 0f, + 1f, -1f, 0f, + 1f, 1f, 0f, + -1f, 1f, 0f, + }; + byte[] colors = + { + 255, 0, 0, 255, + 255, 0, 0, 255, + 255, 0, 0, 255, + 255, 0, 0, 255, + }; + int[] indices = { 0, 1, 2, 0, 2, 3 }; + + fixed (float* p = positions) meshes.Write(mesh, MeshManager.BufferXyz, 0, (IntPtr)p, positions.Length * 4); + fixed (byte* c = colors) meshes.Write(mesh, MeshManager.BufferRgba, 0, (IntPtr)c, colors.Length); + fixed (int* i = indices) meshes.Write(mesh, -1, 0, (IntPtr)i, indices.Length * 4); + + TranslatedProgram translated = ShaderTranslator.Translate(new[] + { + new ShaderStageSource + { + Stage = EnumShaderType.VertexShader, + Filename = "mesh.vsh", + Code = """ + #version 330 core + layout(location = 0) in vec3 position; + layout(location = 1) in vec4 color; + out vec4 vertexColor; + void main(void) + { + gl_Position = vec4(position, 1.0); + vertexColor = color; + } + """, + }, + new ShaderStageSource + { + Stage = EnumShaderType.FragmentShader, + Filename = "mesh.fsh", + Code = """ + #version 330 core + in vec4 vertexColor; + out vec4 outColor; + void main(void) { outColor = vertexColor; } + """, + }, + }, compiler); + Assert.True(translated.Success, string.Join("; ", translated.Errors)); + + using var program = new ShaderProgramResources(context!, 1, translated); + state.SetProgram(1); + + VulkanFramebuffer bound = targets.Get(framebuffer)!; + int formatsId = targets.FormatsIdOf(bound); + RenderTargetFormats formats = targets.FormatsOf(formatsId); + + Pipeline pipeline = pipelines.Get( + state.BuildKey(meshes.LayoutIdOf(mesh), formatsId, 1), + new GraphicsPipelineCache.PipelineRequest + { + Program = program, + VertexLayout = meshes.LayoutOf(meshes.LayoutIdOf(mesh)), + Targets = formats, + Blend = new[] { state.BlendFor(0) }, + PolygonMode = state.PolygonMode, + Topology = state.Topology, + }); + + commands.SubmitAndWait(commandBuffer => + { + targets.Bind(commandBuffer, framebuffer); + targets.EnsureRendering(commandBuffer); + + Vk api = context!.Api; + api.CmdBindPipeline(commandBuffer, PipelineBindPoint.Graphics, pipeline); + + var viewport = new Viewport(0, 0, size, size, 0, 1); + api.CmdSetViewport(commandBuffer, 0, 1, &viewport); + var scissor = new Rect2D(new Offset2D(0, 0), new Extent2D(size, size)); + api.CmdSetScissor(commandBuffer, 0, 1, &scissor); + SetDynamicDefaults(api, commandBuffer); + + meshes.Draw(commandBuffer, mesh); + targets.EndRendering(commandBuffer); + }); + + byte[] pixels = ReadTexture(context!, commands, textures, target, size); + + // The quad covers the target, so the middle is the vertex colour. + int centre = (int)((size / 2 * size + size / 2) * 4); + Assert.Equal(255, pixels[centre + 0]); + Assert.Equal(0, pixels[centre + 1]); + Assert.Equal(255, pixels[centre + 3]); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + private static void SetDynamicDefaults(Vk api, CommandBuffer commandBuffer) + { + api.CmdSetCullMode(commandBuffer, CullModeFlags.None); + api.CmdSetFrontFace(commandBuffer, PipelineKeyState.FrontFace); + api.CmdSetPrimitiveTopology(commandBuffer, PrimitiveTopology.TriangleList); + api.CmdSetDepthTestEnable(commandBuffer, false); + api.CmdSetDepthWriteEnable(commandBuffer, false); + api.CmdSetDepthCompareOp(commandBuffer, CompareOp.Always); + api.CmdSetStencilTestEnable(commandBuffer, false); + api.CmdSetStencilOp(commandBuffer, StencilFaceFlags.FaceFrontAndBack, + StencilOp.Keep, StencilOp.Keep, StencilOp.Keep, CompareOp.Always); + api.CmdSetStencilCompareMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0xFF); + api.CmdSetStencilWriteMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0xFF); + api.CmdSetStencilReference(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0); + api.CmdSetLineWidth(commandBuffer, 1.0f); + } + + [SkippableFact] + public void UpdatingSeparateMeshSlicesPreservesEveryDestinationOffset() + { + using var device = GpuTest.CreateDevice(_output); + const int slices = 3, floatsPerSlice = 9; + int mesh = device.CreateEmptyMesh(slices * floatsPerSlice * sizeof(float), 0, 0, 0, 0, 0, + null, null, null, null, EnumDrawMode.Triangles, staticDraw: false, ssbo: false); + for (int slice = slices - 1; slice >= 0; slice--) + { + float[] values = Enumerable.Range(0, floatsPerSlice) + .Select(index => slice * 100f + index).ToArray(); + device.UpdateMesh(mesh, new MeshData(3, 0) + { + xyz = values, + XyzOffset = slice * floatsPerSlice * sizeof(float), + VerticesCount = 3, + }); + } + var actual = new float[slices * floatsPerSlice]; + Marshal.Copy(device.GetMappedPointer(mesh, EnumMeshBufferPart.Xyz), actual, 0, actual.Length); + for (int slice = 0; slice < slices; slice++) + for (int index = 0; index < floatsPerSlice; index++) + Assert.Equal(slice * 100f + index, actual[slice * floatsPerSlice + index]); + device.DeleteMesh(mesh); + GpuTest.AssertClean(device); + } + + private static unsafe byte[] ReadTexture( + VulkanContext context, SetupQueue commands, TextureManager textures, int textureId, uint size) + { + VulkanTexture texture = textures.Get(textureId)!; + ulong bytes = (ulong)size * size * 4; + + using var readback = new VulkanBuffer(context, bytes, + BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + + commands.SubmitAndWait(commandBuffer => + { + textures.TransitionTexture(commandBuffer, texture, ImageLayout.TransferSrcOptimal); + var region = new BufferImageCopy + { + ImageSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + ImageExtent = new Extent3D(size, size, 1), + }; + context.Api.CmdCopyImageToBuffer(commandBuffer, texture.Image, + ImageLayout.TransferSrcOptimal, readback.Handle, 1, ®ion); + }); + + var result = new byte[(int)bytes]; + Marshal.Copy(readback.Mapped, result, 0, result.Length); + return result; + } + +} +} + +// Source: Optimum.Render.Vulkan.Tests/VertexAttributeDefaultTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; + +/// +/// GL answers a read of a vertex attribute the draw does not supply with the +/// current generic attribute, which defaults to (0, 0, 0, 1). Vulkan has no such +/// thing, and the difference is not academic: the GUI quad carries only positions +/// and UVs while gui.vsh declares six inputs, and gui.fsh discards a fragment +/// based on one of the missing ones. Undefined there meant the entire interface +/// rendered as nothing, with no validation message to say why. +/// +public class VertexAttributeDefaultTests +{ + private static VertexInputSlot Slot(string name, int location, string type) + { + Assert.True(GlslType.TryParse(type, out GlslType parsed), "unknown type " + type); + return new VertexInputSlot(name, location, parsed); + } + + /// The GUI quad's own shape: positions at 0, UVs at 1, nothing else. + private static VertexLayoutDescription QuadLayout() => new( + new[] + { + new VertexBinding(0, 12, PerInstance: false), + new VertexBinding(1, 8, PerInstance: false), + }, + new[] + { + new VertexAttribute(0, 0, Format.R32G32B32Sfloat, 0), + new VertexAttribute(1, 1, Format.R32G32Sfloat, 0), + }); + + [Fact] + public void MissingAttributesGetABindingOfTheirOwn() + { + var declared = new List + { + Slot("vertexPositionIn", 0, "vec3"), + Slot("uvIn", 1, "vec2"), + Slot("colorIn", 2, "vec4"), + Slot("renderFlagsIn", 3, "int"), + Slot("damageEffectIn", 4, "float"), + Slot("jointId", 5, "int"), + }; + + VertexLayoutDescription merged = QuadLayout().WithDefaultsFor(declared); + + Assert.Equal(6, merged.Attributes.Length); + Assert.Equal(3, merged.Bindings.Length); + Assert.Equal(VertexLayoutDescription.DefaultAttributeBinding, merged.Bindings[^1].Binding); + + // Stride zero is what makes every vertex read the same constant. + Assert.Equal(0u, merged.Bindings[^1].Stride); + Assert.False(merged.Bindings[^1].PerInstance); + } + + [Fact] + public void SuppliedAttributesAreLeftOnTheirOwnBinding() + { + var declared = new List + { + Slot("vertexPositionIn", 0, "vec3"), + Slot("uvIn", 1, "vec2"), + Slot("colorIn", 2, "vec4"), + }; + + VertexLayoutDescription merged = QuadLayout().WithDefaultsFor(declared); + + VertexAttribute position = merged.Attributes.Single(a => a.Location == 0); + VertexAttribute uv = merged.Attributes.Single(a => a.Location == 1); + Assert.Equal(0u, position.Binding); + Assert.Equal(1u, uv.Binding); + + VertexAttribute color = merged.Attributes.Single(a => a.Location == 2); + Assert.Equal(VertexLayoutDescription.DefaultAttributeBinding, color.Binding); + } + + /// + /// Integer attributes have to read integer zeros, so they take the second + /// half of the defaults buffer rather than reinterpreting float bits. + /// + [Theory] + [InlineData("float", Format.R32Sfloat, 0u)] + [InlineData("vec2", Format.R32G32Sfloat, 0u)] + [InlineData("vec3", Format.R32G32B32Sfloat, 0u)] + [InlineData("vec4", Format.R32G32B32A32Sfloat, 0u)] + [InlineData("int", Format.R32Sint, 16u)] + [InlineData("ivec4", Format.R32G32B32A32Sint, 16u)] + [InlineData("uint", Format.R32Uint, 16u)] + public void DefaultsUseTheFormatAndHalfMatchingTheDeclaredType( + string type, Format expectedFormat, uint expectedOffset) + { + VertexLayoutDescription merged = + VertexLayoutDescription.Empty.WithDefaultsFor(new[] { Slot("x", 7, type) }); + + VertexAttribute attribute = Assert.Single(merged.Attributes); + Assert.Equal(7u, attribute.Location); + Assert.Equal(expectedFormat, attribute.Format); + Assert.Equal(expectedOffset, attribute.Offset); + } + + [Fact] + public void UnsignedVectorTypesUseUnsignedFormats() + { + VertexLayoutDescription merged = + VertexLayoutDescription.Empty.WithDefaultsFor(new[] { Slot("x", 7, "uvec3") }); + + VertexAttribute attribute = Assert.Single(merged.Attributes); + Assert.Equal(Format.R32G32B32Uint, attribute.Format); + } + + /// + /// A layout that already covers everything must come back untouched, so no + /// pipeline gains a binding it will never have a buffer for. + /// + [Fact] + public void NothingIsAddedWhenTheMeshCoversEveryDeclaredInput() + { + var declared = new List + { + Slot("vertexPositionIn", 0, "vec3"), + Slot("uvIn", 1, "vec2"), + }; + + VertexLayoutDescription layout = QuadLayout(); + VertexLayoutDescription merged = layout.WithDefaultsFor(declared); + + Assert.Same(layout, merged); + } + + /// + /// A program declaring no inputs at all - the fullscreen passes - must not + /// grow a binding either. + /// + [Fact] + public void NothingIsAddedForAProgramWithNoDeclaredInputs() + { + VertexLayoutDescription layout = QuadLayout(); + Assert.Same(layout, layout.WithDefaultsFor(Array.Empty())); + } + + /// + /// The layout the parser produces for gui.vsh is the case this all exists + /// for, so it is pinned against the real shader's declarations rather than a + /// hand-written list. + /// + [Fact] + public void TheGuiShaderDeclaresEveryInputItReads() + { + const string source = """ + #version 330 core + layout(location = 0) in vec3 vertexPositionIn; + layout(location = 1) in vec2 uvIn; + layout(location = 2) in vec4 colorIn; + layout(location = 3) in int renderFlagsIn; + layout(location = 4) in float damageEffectIn; + layout(location = 5) in int jointId; + void main() { gl_Position = vec4(vertexPositionIn, 1.0); } + """; + + ProgramInterfaceLayout layout = ProgramInterfaceLayout.Build( + new[] { (EnumShaderType.VertexShader, GlslParser.Parse(source)) }); + + Assert.Equal(6, layout.VertexInputs.Count); + Assert.Contains(layout.VertexInputs, s => s.Name == "damageEffectIn" && s.Location == 4); + Assert.Contains(layout.VertexInputs, s => s.Name == "renderFlagsIn" && s.Location == 3); + } + + /// + /// Sampler locations have to be tellable apart from uniform block offsets, + /// which start at zero, and from GL's "not found" answer of -1 - otherwise + /// assigning a sampler its texture unit would write into the uniform block + /// at some arbitrary offset instead. + /// + [Theory] + [InlineData(-2, true)] + [InlineData(-3, true)] + [InlineData(-100, true)] + [InlineData(-1, false)] + [InlineData(0, false)] + [InlineData(36, false)] + public void SamplerLocationsAreDisjointFromBlockOffsets(int location, bool isSampler) + { + Assert.Equal(isSampler, ShaderProgramResources.IsSamplerLocation(location)); + } + + /// + /// Asking for Vulkan by name gets it wherever it runs; the automatic setting + /// is a default nobody chose, so it only selects the backend on drivers the + /// backend has been exercised against. + /// + [SkippableFact] + public void AnExplicitRendererChoiceIgnoresTheAutomaticAllowList() + { + Skip.IfNot(VulkanDevice.IsSupported(out string? probeReason), probeReason ?? "No usable Vulkan device."); + + // Explicit: allowed on whatever this machine has. + Assert.True(VulkanDevice.IsSupported(false, out _, out string driver)); + Assert.False(string.IsNullOrWhiteSpace(driver)); + + // A supported software driver can be explicitly selected without being + // on the automatic allow-list (SwiftShader is one such driver). + Assert.Equal(VulkanDevice.IsAllowedForAutomaticSelection(driver), + VulkanDevice.IsSupported(true, out _, out _)); + } + + /// + /// The allow-list itself, which is the part with the logic. Only the drivers + /// the backend is actually run against may be picked automatically; anything + /// unrecognised stays on OpenGL rather than becoming the first person to try + /// the backend on that driver. + /// + [Theory] + [InlineData("NVIDIA", true)] + [InlineData("Intel open-source Mesa driver", true)] + [InlineData("Intel Corporation", true)] + [InlineData("radv", true)] + [InlineData("AMD proprietary driver", true)] + [InlineData("Mesa llvmpipe", true)] + [InlineData("SwiftShader", false)] + [InlineData("MoltenVK", false)] + [InlineData("unknown", false)] + [InlineData("", false)] + public void AutomaticSelectionOnlyTakesKnownDrivers(string driverName, bool allowed) + { + Assert.Equal(allowed, VulkanDevice.IsAllowedForAutomaticSelection(driverName)); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/Rendering/NativeMeshTests.cs b/Optimum.Render.Vulkan.Tests/Rendering/NativeMeshTests.cs new file mode 100644 index 00000000..059f6921 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Rendering/NativeMeshTests.cs @@ -0,0 +1,1072 @@ +// Source: Optimum.Render.Vulkan.Tests/NativeMeshDrawTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Xunit; +using Xunit.Abstractions; + +/// +/// The native device API's mesh draws (docs/vulkan.md, decision 4: +/// "fullscreen triangle, mesh, multi-draw or instanced"), on the real chunk program and a real +/// tesselated face. +/// +/// Stage 1 recorded fullscreen draws only. These pin the four facts a world system depends on: +/// a native mesh draw puts the same pixels on the target as the stated draw of the same mesh +/// with the same state; a mesh pipeline is never the fullscreen pipeline of the same program; +/// a native multi-draw takes its own region of the per-slot indirect ring, the same ring the +/// stated multi-draw allocates from; and the stats count mesh, instanced and indirect draws +/// apart from fullscreen ones. +/// +public class NativeMeshDrawTests(ITestOutputHelper output) +{ + private const int Size = 32; + private const int UpNormalFlags = 7 << 18; + + // ------------------------------------------------------------------- the tests + + /// + /// The same face, drawn twice into the same target: once through the tests' GL-shaped + /// DrawMesh (the platform's generic stated draw) and once through + /// . Both paths run the same shader + /// over the same vertices with the same fixed state, so the pixels are bitwise equal. + /// + [SkippableFact] + public unsafe void ANativeMeshDrawMatchesTheStatedDrawOfTheSameMesh() + { + using Session session = Open(); + + byte[] stated = session.RunStatedFrame(); + + long meshDrawsBefore = session.Device.NativeMeshDrawsForTests; + long fullscreenBefore = session.Device.NativeFullscreenDrawsForTests; + byte[] native = session.RunNativeFrame(); + + Assert.Equal(1, session.Device.NativeMeshDrawsForTests - meshDrawsBefore); + Assert.Equal(0, session.Device.NativeFullscreenDrawsForTests - fullscreenBefore); + + output.WriteLine("stated centre: " + Centre(stated) + " native centre: " + Centre(native)); + Assert.Equal(stated, native); + GpuTest.AssertClean(session.Device); + } + + /// + /// The pipeline key carries the vertex layout, so the mesh pipeline and the fullscreen + /// pipeline of one program, one target and one blend set are two entries, never one. Before + /// the layout entered the key they would have collided and a mesh draw would have run the + /// fullscreen pipeline, which binds no vertex buffers at all. + /// + [SkippableFact] + public void AMeshPipelineKeyNeverCollidesWithAFullscreenOne() + { + using Session session = Open(); + VulkanDevice device = session.Device; + + int before = device.NativePipelinesForTests; + NativePipeline? fullscreen = device.RequestNativePipeline( + session.Description(MeshManager.EmptyLayoutId), out string fullscreenError); + NativePipeline? mesh = device.RequestNativePipeline( + session.Description(session.LayoutId), out string meshError); + + Assert.True(fullscreen != null, fullscreenError); + Assert.True(mesh != null, meshError); + Assert.NotSame(fullscreen, mesh); + Assert.NotEqual(fullscreen!.Key, mesh!.Key); + Assert.Equal(2, device.NativePipelinesForTests - before); + + // Asking again for either gives the entry that was made for it, not the other one. + Assert.Same(mesh, device.RequestNativePipeline(session.Description(session.LayoutId), out _)); + Assert.Same(fullscreen, + device.RequestNativePipeline(session.Description(MeshManager.EmptyLayoutId), out _)); + Assert.Equal(2, device.NativePipelinesForTests - before); + } + + /// + /// The other new state dimensions are in the key too: polygon mode, front face, line width + /// and the bound-depth declaration each make a distinct pipeline, so a cached one is never + /// handed to a draw that asked for different state. + /// + [SkippableFact] + public void EveryNewStateDimensionMakesItsOwnPipeline() + { + using Session session = Open(); + VulkanDevice device = session.Device; + + int before = device.NativePipelinesForTests; + Assert.NotNull(device.RequestNativePipeline(session.Description(session.LayoutId), out _)); + + NativePipelineDescription lines = session.Description(session.LayoutId); + lines.PolygonMode = PolygonMode.Line; + Assert.NotNull(device.RequestNativePipeline(lines, out _)); + + NativePipelineDescription winding = session.Description(session.LayoutId); + winding.FrontFace = FrontFace.CounterClockwise; + Assert.NotNull(device.RequestNativePipeline(winding, out _)); + + NativePipelineDescription wide = session.Description(session.LayoutId); + wide.Topology = PrimitiveTopology.LineList; + wide.LineWidth = 2f; + Assert.NotNull(device.RequestNativePipeline(wide, out _)); + + NativePipelineDescription depthRead = session.Description(session.LayoutId); + depthRead.DepthWrite = false; + depthRead.SamplesBoundDepth = true; + Assert.NotNull(device.RequestNativePipeline(depthRead, out _)); + + Assert.Equal(5, device.NativePipelinesForTests - before); + } + + /// A pipeline that samples the depth it draws into may not also write it. + [SkippableFact] + public void APipelineCannotBothSampleAndWriteTheBoundDepth() + { + using Session session = Open(); + NativePipelineDescription description = session.Description(session.LayoutId); + description.DepthWrite = true; + description.SamplesBoundDepth = true; + + Assert.Null(session.Device.RequestNativePipeline(description, out string error)); + Assert.Contains("cannot also write depth", error, StringComparison.Ordinal); + } + + /// + /// Two native multi-draws in one frame take two regions of the slot's indirect buffer, as + /// the stated route does: writing both at offset zero was the Phase 1B bug where every + /// multi-draw in a frame executed with the ranges of whichever was recorded last. + /// + [SkippableFact] + public unsafe void TwoNativeMultiDrawsTakeTwoRegionsOfTheIndirectRing() + { + using Session session = Open(); + VulkanDevice device = session.Device; + + long indirectBefore = device.NativeIndirectDrawsForTests; + ulong cursorBefore = 0; + ulong cursorAfter = 0; + + session.RunFrame(() => + { + int slot = device.IndirectRingForTests.Current; + cursorBefore = device.IndirectRingForTests.CursorOf(slot); + NativePipeline pipeline = session.BeginNativeSceneDraw(); + Assert.True(device.DrawNativeMeshMulti(pipeline, session.Mesh, + new[] { 0, 0 }, new[] { 6 }, 1, session.Textures(pipeline))); + Assert.True(device.DrawNativeMeshMulti(pipeline, session.Mesh, + new[] { 0, 0 }, new[] { 6 }, 1, session.Textures(pipeline))); + cursorAfter = device.IndirectRingForTests.CursorOf(slot); + device.EndNativePass(); + }); + + Assert.Equal(2, device.NativeIndirectDrawsForTests - indirectBefore); + ulong command = (ulong)sizeof(DrawIndexedIndirectCommand); + Assert.Equal(2 * command, cursorAfter - cursorBefore); + GpuTest.AssertClean(device); + } + + /// + /// An instanced native draw is counted as one, apart from the plain mesh draws, and draws + /// through the mesh's own bindings exactly as + /// does. + /// + [SkippableFact] + public void AnInstancedNativeDrawIsCountedApartFromAPlainMeshDraw() + { + using Session session = Open(); + VulkanDevice device = session.Device; + + long meshBefore = device.NativeMeshDrawsForTests; + long instancedBefore = device.NativeInstancedDrawsForTests; + + session.RunFrame(() => + { + NativePipeline pipeline = session.BeginNativeSceneDraw(); + Assert.True(device.DrawNativeMesh(pipeline, session.Mesh, session.Textures(pipeline))); + Assert.True(device.DrawNativeMeshInstanced(pipeline, session.Mesh, 3, session.Textures(pipeline))); + device.EndNativePass(); + }); + + Assert.Equal(1, device.NativeMeshDrawsForTests - meshBefore); + Assert.Equal(1, device.NativeInstancedDrawsForTests - instancedBefore); + GpuTest.AssertClean(device); + } + + /// + /// A mesh drawn through a pipeline built for another mesh's layout is refused rather than + /// recorded: the vertex buffers it would read are not the ones the pipeline declares, and + /// no validation layer can see that, because every descriptor involved is valid. + /// + [SkippableFact] + public void AMeshDrawThroughTheWrongLayoutsPipelineIsRefused() + { + using Session session = Open(); + VulkanDevice device = session.Device; + + session.RunFrame(() => + { + NativePipeline? fullscreen = device.RequestNativePipeline( + session.Description(MeshManager.EmptyLayoutId), out string error); + Assert.True(fullscreen != null, error); + Assert.True(session.BeginPass()); + Assert.False(device.DrawNativeMesh(fullscreen!, session.Mesh, session.Textures(fullscreen!))); + device.EndNativePass(); + }); + + GpuTest.AssertClean(device); + } + + + [SkippableFact] + public unsafe void IndexedLineStripKeepsItsInteriorEmptyAndDoesNotChangeTheNextMeshTopology() + { + using var device = GpuTest.CreateDevice(output); + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(location = 0) in vec3 position; + void main() { gl_Position = vec4(position, 1); } + """, """ + #version 330 core + out vec4 color; + void main() { color = vec4(1); } + """, "indexed-line-strip"); + var rectangle = new MeshData(4, 5) + { + xyz = new[] { -.75f, -.75f, 0f, .75f, .75f, 0f, + .75f, -.75f, 0f, -.75f, .75f, 0f }, + VerticesCount = 4, + Indices = new[] { 0, 2, 1, 3, 0 }, + IndicesCount = 5, + mode = EnumDrawMode.LineStrip, + }; + int lines = device.CreateMesh(rectangle, true); + int image = device.CreateTexture2D(32, 32, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int target = device.CreateFramebuffer(32, 32); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + + void Begin() + { + device.BeginFrame(); device.BindFramebuffer(target); device.UseProgram(program); + device.SetViewport(0, 0, 32, 32); device.SetDepthTest(false); + device.SetCullFace(false); device.SetBlend(false, EnumBlendMode.Standard); + device.ClearColor(0, 0, 0, 0, 1); + } + byte[] pixels = new byte[32 * 32 * 4]; + void Read() + { + fixed (byte* pointer = pixels) + device.ReadDefaultFramebuffer(0, 0, 32, 32, (IntPtr)pointer); + } + + Begin(); device.DrawMesh(lines); Read(); + Assert.Equal(0, pixels[(16 * 32 + 16) * 4]); + int lit = Enumerable.Range(0, 32 * 32).Count(pixel => pixels[pixel * 4] == 255); + Assert.InRange(lit, 80, 112); + device.Present(); + + rectangle.mode = EnumDrawMode.Triangles; + rectangle.Indices = new[] { 0, 2, 1, 0, 1, 3 }; + rectangle.IndicesCount = 6; + int triangles = device.CreateMesh(rectangle, true); + Begin(); device.DrawMesh(triangles); Read(); + Assert.Equal(255, pixels[(16 * 32 + 16) * 4]); + device.Present(); + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void EveryRequestedInstanceProducesItsOwnPixels() + { + using var device = GpuTest.CreateDevice(output); + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(location = 0) in vec3 position; + void main() { + vec2 offset = vec2(gl_InstanceID) * 0.75; + gl_Position = vec4(position.xy + offset, position.z, 1); + } + """, """ + #version 330 core + out vec4 color; + void main() { color = vec4(1, 0, 1, 1); } + """, "instance-pixels"); + var quad = new MeshData(4, 6) + { + xyz = new[] { -1f, -1f, 0f, -.5f, -1f, 0f, + -.5f, -.5f, 0f, -1f, -.5f, 0f }, + VerticesCount = 4, + Indices = new[] { 0, 1, 2, 0, 2, 3 }, + IndicesCount = 6, + mode = EnumDrawMode.Triangles, + }; + int mesh = device.CreateMesh(quad, true); + int image = device.CreateTexture2D(16, 16, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int target = device.CreateFramebuffer(16, 16); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + device.BeginFrame(); device.BindFramebuffer(target); device.UseProgram(program); + device.SetViewport(0, 0, 16, 16); device.SetDepthTest(false); + device.SetCullFace(false); device.SetBlend(false, EnumBlendMode.Standard); + device.ClearColor(0, 0, 0, 0, 1); + device.DrawMeshInstanced(mesh, 2); + byte[] pixels = new byte[16 * 16 * 4]; + fixed (byte* pointer = pixels) + device.ReadDefaultFramebuffer(0, 0, 16, 16, (IntPtr)pointer); + foreach ((int x, int y) in new[] { (2, 2), (8, 8) }) + { + int pixel = (y * 16 + x) * 4; + Assert.Equal(new byte[] { 255, 0, 255, 255 }, pixels[pixel..(pixel + 4)]); + } + device.Present(); + GpuTest.AssertClean(device); + } + + /// The client's IShader, as much of it as CompileShader reads. + private sealed class CorpusShader : IShader + { + public EnumShaderType Type { get; set; } + public string Code { get; set; } = ""; + public string PrefixCode { get; set; } = ""; + public bool Compile() => true; + } + + /// The client's IShaderProgram, as much of it as LinkProgram reads. + private sealed class CorpusProgram : IShaderProgram + { + public int ProgramId { get; set; } + public string AssetDomain { get; set; } = "game"; + public int PassId { get; set; } + public string PassName { get; set; } = "chunkopaque"; + public bool ClampTexturesToEdge { get; set; } + public IShader VertexShader { get; set; } = null!; + public IShader FragmentShader { get; set; } = null!; + public IShader GeometryShader { get; set; } = null!; + public bool Oit { get; set; } = true; + public bool Disposed => false; + public bool LoadError => false; + public Vintagestory.API.Datastructures.OrderedDictionary UBOs { get; } = new(); + public bool Compile() => true; + public bool HasUniform(string uniformName) => false; + public void Use() { } + public void Stop() { } + public void Dispose() { } + public void Uniform(string uniformName, float value) { } + public void Uniform(string uniformName, int value) { } + public void Uniform(string uniformName, Vec2f value) { } + public void Uniform(string uniformName, Vec2i value) { } + public void Uniform(string uniformName, float valueX, float valueY) { } + public void Uniform(string uniformName, Vec3f value) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ, float valueW) { } + public void Uniform(string uniformName, Vec4f value) { } + public void Uniforms4(string uniformName, int count, float[] values) { } + public void UniformMatrix(string uniformName, float[] matrix) { } + public void BindTexture2D(string samplerName, int textureId, int textureNumber) { } + public void BindTextureCube(string samplerName, int textureId, int textureNumber) { } + public void UniformMatrices(string uniformName, int count, float[] matrix) { } + public void UniformMatrices4x3(string uniformName, int count, float[] matrix) { } + } + + // ---------------------------------------------------------------------- session + + private Session Open() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + Skip.IfNot(GpuTest.TryCreateDevice(output, out VulkanDevice? device), "No usable Vulkan device."); + return new Session(device!); + } + + private static string Centre(byte[] pixels) + { + int i = (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + /// + /// The real chunkopaque program, one tesselated face, and a target to draw it into: the + /// smallest setting in which both routes can draw the same thing. + /// + private sealed class Session : IDisposable + { + private readonly Dictionary _samplerTextures = new(StringComparer.Ordinal); + + public Session(VulkanDevice device) + { + Device = device; + Program = LinkChunkOpaque(device); + BindEveryDeclaredSampler(); + + Color = device.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + Depth = device.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, + IntPtr.Zero, false); + Framebuffer = device.CreateFramebuffer(Size, Size); + device.AttachTexture(Framebuffer, EnumFramebufferAttachment.ColorAttachment0, Color, 0); + device.AttachTexture(Framebuffer, EnumFramebufferAttachment.DepthAttachment, Depth, 0); + device.SetDrawBuffers(Framebuffer, 0b1); + Assert.True(device.CheckFramebufferComplete(Framebuffer, out string status), status); + + Mesh = device.CreateMesh(BuildBlockFace(), staticDraw: true); + Assert.True(Mesh > 0, device.GetError() ?? "mesh upload failed"); + LayoutId = device.NativeMeshLayoutId(Mesh); + Assert.True(LayoutId > 0, "the face's vertex layout is the reserved empty one"); + } + + public VulkanDevice Device { get; } + public int Program { get; } + public int Framebuffer { get; } + public int Color { get; } + public int Depth { get; } + public int Mesh { get; } + public int LayoutId { get; } + + public void Dispose() => Device.Dispose(); + + /// The fixed state both routes draw with: depth LEQUAL, writes on, no culling, no blending. + public NativePipelineDescription Description(int layoutId) => new() + { + ProgramId = Program, + Blend = new[] { AttachmentBlend.Default }, + DepthTest = true, + DepthWrite = true, + DepthCompare = CompareOp.LessOrEqual, + Cull = CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + VertexLayoutId = layoutId, + Targets = Device.NativeTargetFormats(Framebuffer, 1u)!, + }; + + /// Every sampler the program declares, at the slot this pipeline resolved for it. + public NativeTexture[] Textures(NativePipeline pipeline) + { + var textures = new List(); + foreach (KeyValuePair sampler in _samplerTextures) + { + NativeSamplerSlot slot = pipeline.Sampler(sampler.Key); + if (slot.IsPresent) textures.Add(new NativeTexture(slot, sampler.Value)); + } + return textures.ToArray(); + } + + public bool BeginPass() => Device.BeginNativePass(new NativePassDescription + { + Name = "NativeMeshDrawTests", + FramebufferId = Framebuffer, + ColorSlots = 1u, + Reads = _samplerTextures.Values.ToArray(), + ViewportWidth = Size, + ViewportHeight = Size, + }); + + /// Opens the pass and returns the pipeline the mesh draws through. + public NativePipeline BeginNativeSceneDraw() + { + NativePipeline? pipeline = Device.RequestNativePipeline(Description(LayoutId), out string error); + Assert.True(pipeline != null, error); + Assert.True(BeginPass()); + return pipeline!; + } + + /// One frame: clear, set the uniforms both routes read, run , present. + public void RunFrame(Action body) + { + Device.BeginFrame(); + Device.BindFramebuffer(Framebuffer); + Device.ClearColor(0, 1f, 0f, 1f, 1f); + Device.ClearDepth(1f); + SetUniforms(); + Device.SetViewport(0, 0, Size, Size); + body(); + Device.Present(); + } + + /// The stated route: the GL-shaped state, then DrawMesh. + public unsafe byte[] RunStatedFrame() + { + byte[] pixels = new byte[Size * Size * 4]; + RunFrame(() => + { + Device.UseProgram(Program); + Device.SetDepthTest(true); + Device.SetDepthMask(true); + Device.SetDepthFunc(0x203); // GL_LEQUAL + Device.SetCullFace(false); + Device.SetBlend(false, EnumBlendMode.Standard); + Device.DrawMesh(Mesh); + }); + Read(pixels); + return pixels; + } + + /// The native route: a declared pass, a pipeline with the mesh's layout, one draw. + public unsafe byte[] RunNativeFrame() + { + byte[] pixels = new byte[Size * Size * 4]; + RunFrame(() => + { + NativePipeline pipeline = BeginNativeSceneDraw(); + Assert.True(Device.DrawNativeMesh(pipeline, Mesh, Textures(pipeline))); + Device.EndNativePass(); + }); + Read(pixels); + return pixels; + } + + private unsafe void Read(byte[] pixels) + { + fixed (byte* destination = pixels) + { + Device.BindFramebuffer(Framebuffer); + Device.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + } + + // ------------------------------------------------------------- program setup + + private void SetUniforms() + { + float[] identity = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + foreach (string name in new[] { "projectionMatrix", "modelViewMatrix", "modelMatrix", "mvpMatrix", + "toShadowMapSpaceMatrixFar", "toShadowMapSpaceMatrixNear" }) + { + int location = Device.GetUniformLocation(Program, name); + if (location >= 0) Device.SetUniformMatrix(Program, location, identity); + } + SetFloat("viewDistance", 1024f); + SetFloat("viewDistanceLod0", 1024f); + SetFloat("alphaTest", 0.001f); + SetFloat("zNear", 0.1f); + SetFloat("zFar", 1024f); + SetFloat("shadowRangeFar", 1024f); + SetFloat("shadowRangeNear", 64f); + SetFloat("shadowMapWidthInv", 1f); + SetFloat("shadowMapHeightInv", 1f); + int ambient = Device.GetUniformLocation(Program, "rgbaAmbientIn"); + if (ambient >= 0) Device.SetUniform(Program, ambient, 1f, 1f, 1f); + int frameSize = Device.GetUniformLocation(Program, "frameSize"); + if (frameSize >= 0) Device.SetUniform(Program, frameSize, (float)Size, (float)Size); + } + + private void SetFloat(string name, float value) + { + int location = Device.GetUniformLocation(Program, name); + if (location >= 0) Device.SetUniform(Program, location, value); + } + + private unsafe void BindEveryDeclaredSampler() + { + var white = new byte[] { 255, 255, 255, 255 }; + int unit = 0; + foreach (string name in Device.SamplerNamesOf(Program)) + { + int texture; + fixed (byte* pixels = white) + { + texture = Device.CreateTexture2D(1, 1, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + } + Device.SetSamplerUnit(Program, name, unit); + Device.BindTexture(unit, texture); + _samplerTextures[name] = texture; + unit++; + } + } + + private static int LinkChunkOpaque(VulkanDevice device) + { + List stages = ShaderCorpus.BuildProgram("chunkopaque", + ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), ShaderCorpus.Variants().First()); + Assert.NotEmpty(stages); + + var program = new CorpusProgram { PassName = "chunkopaque" }; + foreach (ShaderStageSource stage in stages) + { + var shader = new CorpusShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode ?? "", + }; + Assert.True(device.CompileShader(shader), device.GetError() ?? "compile failed"); + if (stage.Stage == EnumShaderType.VertexShader) program.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) program.FragmentShader = shader; + else program.GeometryShader = shader; + } + + int id = device.LinkProgram(program); + Assert.True(id > 0, device.GetError() ?? "link failed"); + return id; + } + + /// One tesselated block face, in the layout the chunk tesselator emits. + private static MeshData BuildBlockFace() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + float[] positions = + { + -0.5f, -0.5f, 0f, + 0.5f, -0.5f, 0f, + 0.5f, 0.5f, 0f, + -0.5f, 0.5f, 0f, + }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags( + positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb, flags: UpNormalFlags); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/NativeStatedTests.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Platform; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +using LinkedProgram = Optimum.Render.Vulkan.Tests.GpuTest.TestProgram; +using LinkedShader = Optimum.Render.Vulkan.Tests.GpuTest.TestShader; + +/// +/// The generic native draw (VulkanClientPlatform.NativeStated.cs, Platform/StatedDraw.cs) against a +/// native draw whose pipeline and pass are written out by hand, on one device: the same mesh, the +/// same program. The generic route gets its fixed state only through the platform's own virtuals; +/// the reference states what OpenGL would do with those calls. Identical pixels mean the record +/// states what the client said. +/// +/// The program is a gui program that is not the registered ShaderPrograms.Gui, so no dedicated +/// route takes the draw: it is exactly the shape of a mod renderer's draw. +/// +public class NativeStatedTests(ITestOutputHelper output) +{ + private const int Size = 16; + + private sealed class StatedPlatform : VulkanClientPlatform + { + public StatedPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + public enum Mask { All, RedGreen } + + /// + /// Blend off, three blend modes, a colour mask and a scissor rectangle: the generic route + /// draws what the hand-stated reference draws, and records one native draw. + /// + [SkippableTheory] + [InlineData(false, EnumBlendMode.Standard, Mask.All, false)] + [InlineData(true, EnumBlendMode.Standard, Mask.All, false)] + [InlineData(true, EnumBlendMode.PremultipliedAlpha, Mask.All, false)] + [InlineData(true, EnumBlendMode.Brighten, Mask.All, false)] + [InlineData(true, EnumBlendMode.Standard, Mask.RedGreen, false)] + [InlineData(true, EnumBlendMode.Standard, Mask.All, true)] + public unsafe void TheStatedRouteDrawsWhatTheHandStatedReferenceDraws(bool blend, EnumBlendMode mode, Mask mask, bool scissor) + { + using Session session = Open(); + + byte[] reference = session.Run(stated: false, blend, mode, mask, scissor); + + long statedBefore = session.Platform.StatedDrawsForTests; + byte[] native = session.Run(stated: true, blend, mode, mask, scissor); + + Assert.Equal(1, session.Platform.StatedDrawsForTests - statedBefore); + output.WriteLine("centre reference " + Centre(reference, 0) + " stated " + Centre(native, 0)); + Assert.Equal(reference, native); + Assert.NotEqual("0,51,102,153", Centre(native, 0)); + GpuTest.AssertClean(session.Seam); + } + + /// + /// A target with two colour attachments and only the first selected as a draw buffer: the + /// second keeps its clear colour - the stated draw buffers are write masks - and the first + /// matches the reference, which writes the first slot only. + /// + [SkippableFact] + public unsafe void AnUnselectedDrawBufferKeepsItsContentsOnBothRoutes() + { + using Session session = Open(); + + (byte[] firstReference, byte[] secondReference) = session.RunTwoTargets(stated: false); + (byte[] firstStated, byte[] secondStated) = session.RunTwoTargets(stated: true); + + Assert.Equal(firstReference, firstStated); + Assert.Equal(secondReference, secondStated); + // The second attachment is still the clear colour (0, 51, 102, 153). + Assert.Equal("0,51,102,153", Centre(secondStated, 0)); + GpuTest.AssertClean(session.Seam); + } + + /// + /// A platform bind is the latest bind: a fork renderer's raw framebuffer bind before it no + /// longer addresses the generic draws and clears (GL has one binding point). + /// + [Fact] + public void APlatformBindReplacesAForkBind() + { + var platform = new StatedPlatform(); + var target = new FrameBufferRef { FboId = 7, Width = Size, Height = Size, ColorTextureIds = new[] { 1 } }; + + platform.NoteForkFramebuffer(12); + Assert.Equal(12, platform.CurrentTargetId); + + platform.CurrentFrameBuffer = target; + Assert.Equal(7, platform.CurrentTargetId); + + platform.NoteForkFramebuffer(12); + platform.BindCurrentFrameBufferKeepViewport(target); + Assert.Equal(7, platform.CurrentTargetId); + } + + private static string Centre(byte[] pixels, int offset) + { + int i = offset + (Size / 2 * Size + Size / 2) * 4; + return pixels[i] + "," + pixels[i + 1] + "," + pixels[i + 2] + "," + pixels[i + 3]; + } + + private Session Open() + { + (string manifest, string reason) = NativeManifest.Value; + Skip.If(manifest.Length == 0, reason); + Session? session = Session.TryOpen(output, manifest); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + private sealed class Session : IDisposable + { + private static readonly float[] Identity = + { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + + public StatedPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + private FrameBufferRef target = null!; + private FrameBufferRef twoTargets = null!; + private MeshRef quad = null!; + private ShaderProgram gui = null!; + private int texture; + private ClientPlatformAbstract? previousPlatform; + private string dataPath = ""; + + public static Session? TryOpen(ITestOutputHelper output, string manifestDirectory) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-native-stated-" + Guid.NewGuid().ToString("N")); + var platform = new StatedPlatform + { + DeviceFactory = () => + { + VulkanDevice created = GpuTest.NewDevice(); + created.NativeShaderDirectory = manifestDirectory; + created.NativeShadersEnabled = true; + created.IgnoreModShaderScan = true; + return created; + }, + CrashMarkerDataPath = dataPath, + }; + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + platform.NativeGuiEnabled = false; + + var session = new Session { Platform = platform, previousPlatform = ScreenManager.Platform, dataPath = dataPath }; + ScreenManager.Platform = platform; + platform.ShaderUniforms = new DefaultShaderUniforms(); + + VulkanDevice seam = platform.GraphicsDevice!; + session.target = CreateTarget(seam, 1); + session.twoTargets = CreateTarget(seam, 2); + InstallFrameBuffers(platform, session.target); + // The draw buffers each target writes, stated the way the platform states its own. + platform.StateDrawBuffers(session.target.FboId, 1); + platform.StateDrawBuffers(session.twoTargets.FboId, 1); + + var program = new ShaderProgram { PassName = "gui" }; + Link(seam, program, "gui", + new[] { "projectionMatrix", "modelViewMatrix", "rgbaIn", "noTexture", "applyColor", "alphaTest" }); + session.gui = program; + session.texture = Gradient(seam); + session.quad = platform.UploadMesh(BuildQuad()); + return session; + } + + public void Dispose() + { + ShaderProgramBase.CurrentShaderProgram = null; + if (quad != null) Platform.DeleteMesh(quad); + ScreenManager.Platform = previousPlatform!; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + + /// + /// One frame: the target bound and cleared, one quad - through the platform with the state + /// set through its virtuals, or ( false) the hand-stated reference. + /// + public unsafe byte[] Run(bool stated, bool blend, EnumBlendMode mode, Mask mask, bool scissor) + { + Platform.BeginFrame(); + Prepare(target); + if (stated) + { + Platform.GlToggleBlend(blend, mode); + if (mask == Mask.RedGreen) Platform.GlColorMask(true, true, false, false); + if (scissor) + { + Platform.GlScissorFlag(true); + Platform.GlScissor(4, 4, 8, 8); + } + Platform.RenderMesh(quad); + } + else + { + AttachmentBlend attachment = AttachmentBlend.For(blend, mode); + if (mask == Mask.RedGreen) attachment.WriteMask = ColorComponentFlags.RBit | ColorComponentFlags.GBit; + DrawReference(target, attachment, + scissor ? new Rect2D(new Offset2D(4, 4), new Extent2D(8, 8)) : null); + } + + Platform.GlColorMask(true, true, true, true); + Platform.GlScissorFlag(false); + byte[] pixels = Read(target.ColorTextureIds[0]); + Platform.EndFrame(); + return pixels; + } + + public unsafe (byte[] First, byte[] Second) RunTwoTargets(bool stated) + { + Platform.BeginFrame(); + Prepare(twoTargets); + if (stated) + { + Platform.GlToggleBlend(false); + Platform.RenderMesh(quad); + } + else + { + DrawReference(twoTargets, AttachmentBlend.For(false, EnumBlendMode.Standard), null); + } + byte[] first = Read(twoTargets.ColorTextureIds[0]); + byte[] second = Read(twoTargets.ColorTextureIds[1]); + Platform.EndFrame(); + return (first, second); + } + + private void Prepare(FrameBufferRef frameBuffer) + { + VulkanDevice seam = Seam; + // Clears are not part of the comparison: every attachment starts from the same colour. + seam.BindFramebuffer(frameBuffer.FboId); + seam.SetDrawBuffers(frameBuffer.FboId, (1 << frameBuffer.ColorTextureIds.Length) - 1); + for (int i = 0; i < frameBuffer.ColorTextureIds.Length; i++) seam.ClearColor(i, 0f, 0.2f, 0.4f, 0.6f); + Platform.StateDrawBuffers(frameBuffer.FboId, 1); + + Platform.CurrentFrameBuffer = frameBuffer; + Platform.GlDisableDepthTest(); + Platform.GlDepthMask(false); + Platform.GlDisableCullFace(); + + Platform.UseShaderProgram(gui.ProgramId); + ShaderProgramBase.CurrentShaderProgram = gui; + seam.SetUniformMatrix(gui.ProgramId, gui.uniformLocations["projectionMatrix"], Identity); + seam.SetUniformMatrix(gui.ProgramId, gui.uniformLocations["modelViewMatrix"], Identity); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["rgbaIn"], 1f, 0.75f, 0.5f, 0.8f); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["noTexture"], 0f); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["applyColor"], 1); + seam.SetUniform(gui.ProgramId, gui.uniformLocations["alphaTest"], 0f); + Platform.BindProgramTexture2D(gui, "tex2d", texture, 0); + Platform.BindProgramTexture2D(gui, "tex2dOverlay", 0, 1); + } + + /// + /// The quad drawn with everything written out: depth and cull off, slot 0 only, the + /// full-target viewport, the program's two samplers on the gradient and on nothing. + /// + private void DrawReference(FrameBufferRef frameBuffer, AttachmentBlend attachment, Rect2D? scissor) + { + VulkanDevice seam = Seam; + int meshId = ((VAO)quad).VaoId; + NativePipeline? pipeline = seam.RequestNativePipeline(new NativePipelineDescription + { + ProgramId = gui.ProgramId, + Blend = new[] { attachment }, + DepthTest = false, + DepthWrite = false, + Cull = CullModeFlags.None, + Topology = seam.NativeMeshTopology(meshId), + VertexLayoutId = seam.NativeMeshLayoutId(meshId), + Targets = seam.NativeTargetFormats(frameBuffer.FboId, 1u)!, + }, out string error); + Assert.True(pipeline != null, error); + Assert.True(seam.BeginNativePass(new NativePassDescription + { + Name = "Reference", + FramebufferId = frameBuffer.FboId, + ColorSlots = 1u, + Reads = new[] { texture }, + Scissor = scissor, + })); + Assert.True(seam.DrawNativeMesh(pipeline!, meshId, new[] + { + new NativeTexture(pipeline!.Sampler("tex2d"), texture), + new NativeTexture(pipeline.Sampler("tex2dOverlay"), 0), + })); + seam.EndNativePass(); + } + + private unsafe byte[] Read(int textureId) + { + VulkanDevice seam = Seam; + int reader = seam.CreateFramebuffer(Size, Size); + seam.AttachTexture(reader, EnumFramebufferAttachment.ColorAttachment0, textureId, 0); + seam.SetDrawBuffers(reader, 1); + seam.BindFramebuffer(reader); + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadDefaultFramebuffer(0, 0, Size, Size, (IntPtr)destination); + } + seam.DeleteFramebuffer(reader); + return pixels; + } + + private static void Link(VulkanDevice seam, ShaderProgramBase program, string name, string[] uniforms) + { + List stages = ShaderCorpus.BuildProgram( + name, ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), new ShaderCorpus.ShaderVariant()); + var linked = new LinkedProgram { PassName = name }; + foreach (ShaderStageSource stage in stages) + { + var shader = new LinkedShader { Type = stage.Stage, Code = stage.Code, PrefixCode = stage.PrefixCode }; + Assert.True(seam.CompileShader(shader)); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + int id = seam.LinkProgram(linked); + Assert.True(id > 0, seam.GetError() ?? "link failed"); + program.ProgramId = id; + foreach (string uniform in uniforms) + { + int location = seam.GetUniformLocation(id, uniform); + Assert.True(location != -1, name + " has no location for " + uniform); + program.uniformLocations[uniform] = location; + } + } + + private static FrameBufferRef CreateTarget(VulkanDevice seam, int attachments) + { + var target = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new int[attachments], + }; + for (int i = 0; i < attachments; i++) + { + target.ColorTextureIds[i] = seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + seam.AttachTexture(target.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + i), + target.ColorTextureIds[i], 0); + } + seam.SetDrawBuffers(target.FboId, (1 << attachments) - 1); + Assert.True(seam.CheckFramebufferComplete(target.FboId, out string status), status); + return target; + } + + private static void InstallFrameBuffers(StatedPlatform platform, FrameBufferRef target) + { + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = target; + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + } + + private static unsafe int Gradient(VulkanDevice seam) + { + var pixels = new byte[8 * 8 * 4]; + for (int y = 0; y < 8; y++) + for (int x = 0; x < 8; x++) + { + int i = (y * 8 + x) * 4; + pixels[i] = (byte)(16 + x * 30); + pixels[i + 1] = (byte)(32 + y * 25); + pixels[i + 2] = (byte)(((x + y) & 1) * 200 + 20); + pixels[i + 3] = (byte)(96 + ((x + y) & 3) * 40); + } + fixed (byte* first = pixels) + { + return seam.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)first, false); + } + } + + private static MeshData BuildQuad() + { + var mesh = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: false); + float[] positions = { -1f, -1f, 0f, 1f, -1f, 0f, 1f, 1f, 0f, -1f, 1f, 0f }; + float[] uvs = { 0f, 0f, 1f, 0f, 1f, 1f, 0f, 1f }; + for (int i = 0; i < 4; i++) + { + mesh.AddVertexWithFlags(positions[i * 3], positions[i * 3 + 1], positions[i * 3 + 2], + uvs[i * 2], uvs[i * 2 + 1], ColorUtil.WhiteArgb, 0); + } + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) mesh.AddIndex(index); + return mesh; + } + } + + private static readonly Lazy<(string Directory, string Reason)> NativeManifest = new(BuildNativeShaders); + + private static (string, string) BuildNativeShaders() + { + if (!NativeShaderTree.TryCreateCompiler(out ShaderCompiler? compiler, out string reason)) return ("", reason); + using (compiler) + { + var builder = new NativeShaderBuilder(compiler!); + NativeShaderBuildResult result = builder.Build(Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"), "gui"); + if (!result.Success) return ("", string.Join("\n", result.Errors)); + string root = Path.Combine(Path.GetTempPath(), "optimum-native-stated-shaders-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(root); + NativeShaderBuilder.Write(result, root); + return (Path.Combine(root, NativeShaderManifest.DirectoryName), ""); + } + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/ResourceLifetimeTests.cs b/Optimum.Render.Vulkan.Tests/ResourceLifetimeTests.cs new file mode 100644 index 00000000..114f23e7 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ResourceLifetimeTests.cs @@ -0,0 +1,617 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class ResourceLifetimeTests(ITestOutputHelper output) +{ + private const string Triangle = """ + #version 330 core + void main() { + gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); + } + """; + + private VulkanDevice Open(bool aliasing = false) => + GpuTest.CreateDevice(output, device => device.TransientAliasingOverride = aliasing); + + private static int Attach(VulkanDevice device, int texture, int width, int height) + { + int target = device.CreateFramebuffer(width, height); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + device.SetDrawBuffers(target, 1); + return target; + } + + private static void Draw(VulkanDevice device, int framebuffer, int program, int width, int height) + { + device.BindFramebuffer(framebuffer); + device.SetViewport(0, 0, width, height); + device.SetDepthTest(false); + device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.UseProgram(program); + device.DrawFullscreenTriangle(); + } + + private static unsafe byte[] Read(VulkanDevice device, int width, int height) + { + var pixels = new byte[width * height * 4]; + fixed (byte* pointer = pixels) + device.ReadDefaultFramebuffer(0, 0, width, height, (IntPtr)pointer); + return pixels; + } + + [Fact] + public void ResourceAgeWindowExpiresWithoutAffectingPermanentBindings() + { + var age = new ResourceAge(2); + age.NoteFrame(100); age.NoteFrame(200); age.NoteFrame(300); + Assert.False(age.IsShortLived(0)); Assert.False(age.IsShortLived(200)); + Assert.True(age.IsShortLived(201)); + var permanent = new BufferBindingValue(0, new Silk.NET.Vulkan.Buffer(1), 0, 64); + var recent = new SamplerBindingValue(0, new ImageView(1), new Sampler(2), Resource: 201); + var contents = new DescriptorSetContents(1, 0, new[] { recent }, new[] { permanent }); + Assert.True(age.NamesShortLived(contents)); + age.NoteFrame(400); + Assert.False(age.NamesShortLived(contents)); + age.ShortLivedFrames = 0; + Assert.False(age.IsShortLived(401)); + } + + [SkippableTheory] + [InlineData(false, false)] + [InlineData(false, true)] + [InlineData(true, false)] + [InlineData(true, true)] + public void ReusedPostTargetsContainTheCurrentFrame(bool aliasing, bool frameGraph) + { + using var device = Open(aliasing); + device.FrameGraphForTests.Enabled = frameGraph; + int fill = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform float frameBlue; + out vec4 color; + void main() { color = vec4(floor(gl_FragCoord.xy) / 16.0, frameBlue, 1); } + """, "reuse-fill"); + int rotate = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D source; + out vec4 color; + void main() { + vec3 pixel = texelFetch(source, ivec2(gl_FragCoord.xy), 0).rgb; + color = vec4(pixel.g, pixel.b, pixel.r * 0.5 + 0.25, 1); + } + """, "reuse-transform"); + device.SetSamplerUnit(rotate, "source", 0); + int blue = device.GetUniformLocation(fill, "frameBlue"); + int[] textures = new int[4], targets = new int[4]; + for (int i = 0; i < 4; i++) + { + textures[i] = i < 3 + ? device.CreateTransientTexture2D(16, 16, EnumTextureInternalFormat.Rgba8, 2 + i) + : device.CreateTexture2D(16, 16, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + targets[i] = Attach(device, textures[i], 16, 16); + } + byte[] pixels = Array.Empty(); + for (int frame = 0; frame < 5; frame++) + { + device.BeginFrame(); + for (int i = 0; i < 3; i++) device.BindTransientForFrame(textures[i], i, i + 1); + Assert.Equal(aliasing ? 1 : 0, device.Transients.AliasedLeaseCount); + device.SetUniform(fill, blue, (frame + 1) / 16f); + for (int pass = 0; pass < 4; pass++) + { + device.BindTexture(0, pass == 0 ? 0 : textures[pass - 1]); + Draw(device, targets[pass], pass == 0 ? fill : rotate, 16, 16); + } + device.BindTexture(0, 0); + if (frame == 4) pixels = Read(device, 16, 16); + device.Present(); + } + // Three channel rotations leave each original channel halved plus 0.25. + // Allow two UNORM8 rounding steps; expected values come from the scene inputs. + for (int y = 0; y < 16; y++) + for (int x = 0; x < 16; x++) + { + int index = (y * 16 + x) * 4; + Assert.InRange((int)pixels[index], (int)Math.Round((x / 32.0 + .25) * 255) - 2, (int)Math.Round((x / 32.0 + .25) * 255) + 2); + Assert.InRange((int)pixels[index + 1], (int)Math.Round((y / 32.0 + .25) * 255) - 2, (int)Math.Round((y / 32.0 + .25) * 255) + 2); + Assert.InRange((int)pixels[index + 2], 102, 106); + Assert.Equal(255, pixels[index + 3]); + } + if (aliasing) Assert.Equal(2, device.Transients.PhysicalImageCount); + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void SamplingTheTargetUsesAFreshSnapshotForEveryDraw() + { + using var device = Open(); + int swap = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D source; + out vec4 color; + void main() { color = texelFetch(source, ivec2(gl_FragCoord.x < 1.0 ? 1 : 0, 0), 0); } + """, "feedback-swap"); + byte[] original = { 255, 0, 0, 255, 0, 255, 0, 255 }; + int texture; + fixed (byte* pixels = original) + texture = device.CreateTexture2D(2, 1, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + int target = Attach(device, texture, 2, 1); + device.SetSamplerUnit(swap, "source", 0); + for (int frame = 0; frame < 6; frame++) + { + device.BeginFrame(); + device.BindTexture(0, texture); + Draw(device, target, swap, 2, 1); + Draw(device, target, swap, 2, 1); + Assert.Equal(original, Read(device, 2, 1)); + device.Present(); + } + Assert.InRange(device.ReadSelfCopiesForTests.Created, 1, 3); + GpuTest.AssertClean(device); + } + + private sealed class Clock : ITimelineClock + { + public ulong FrameRecorded { get; set; } + public ulong TransferRecorded { get; set; } + public ulong FrameCompleted { get; set; } + public ulong TransferCompleted { get; set; } + } + + [Fact] + public void FeedbackCopiesWaitForCompletionBeforeReuseOrRetirement() + { + var clock = new Clock { FrameRecorded = 7 }; + int next = 0; + var destroyed = new List(); + var pool = new FeedbackCopyPool(clock, _ => ++next, destroyed.Add, idleFrames: 2); + var shape = new FeedbackCopyDesc(16, 16, Format.R8G8B8A8Unorm, 1, 1, false); + int first = pool.Acquire(shape); + pool.Release(first); pool.EndFrame(); pool.Collect(); + int second = pool.Acquire(shape); + Assert.NotEqual(first, second); + Assert.Empty(destroyed); + clock.FrameCompleted = 7; pool.Collect(); + Assert.Equal(first, pool.Acquire(shape)); + pool.Release(first); pool.Release(second); pool.EndFrame(); + for (int i = 0; i < 5; i++) pool.Collect(); + Assert.Equal(0, pool.Live); + Assert.Contains(first, destroyed); + Assert.Contains(second, destroyed); + } + + [SkippableFact] + public void DeletingASharedAttachmentTwiceRetiresItOnce() + { + using var device = Open(); + int image = device.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int before = device.TexturesForTests.Count; + device.DeleteTexture(image); + Assert.Equal(before - 1, device.TexturesForTests.Count); + device.DeleteTexture(image); + Assert.Equal(before - 1, device.TexturesForTests.Count); + Assert.Null(device.TexturesForTests.Get(image)); + device.BeginFrame(); + device.Present(); + GpuTest.AssertClean(device); + } + private sealed class Retirement(Action dispose) : IDisposable + { + public void Dispose() => dispose(); + } + + [Fact] + public void RetirementSnapshotsBothTimelinesAndDoesNotBlockReadyFollowers() + { + var clock = new Clock { FrameRecorded = 9, TransferRecorded = 3 }; + var queue = new RetireQueue(clock); + var destroyed = new List(); + queue.Retire(new Retirement(() => destroyed.Add(1))); + clock.FrameRecorded = 4; clock.TransferRecorded = 8; + queue.Retire(new Retirement(() => destroyed.Add(2))); + clock.FrameRecorded = 100; clock.TransferRecorded = 100; + clock.FrameCompleted = 4; clock.TransferCompleted = 7; + Assert.Equal(0, queue.Collect()); + clock.TransferCompleted = 8; + Assert.Equal(1, queue.Collect()); + Assert.Equal(new[] { 2 }, destroyed); + clock.FrameCompleted = 9; + Assert.Equal(1, queue.Collect()); + Assert.Equal(new[] { 2, 1 }, destroyed); + Assert.Equal(0, queue.Collect()); + Assert.Equal(0, queue.PendingCount); + } + + [Fact] + public void ConcurrentRetirementAndReentrantDisposalDoNotLoseResources() + { + var clock = new Clock { FrameRecorded = 3, TransferRecorded = 5 }; + var queue = new RetireQueue(clock); + var destroyed = new int[128]; + Parallel.For(0, destroyed.Length, i => queue.Retire(new Retirement(() => destroyed[i]++))); + Assert.Equal(destroyed.Length, queue.PendingCount); + Assert.Equal(0, queue.Collect()); + clock.FrameCompleted = 3; clock.TransferCompleted = 5; + Assert.Equal(destroyed.Length, queue.Collect()); + Assert.All(destroyed, count => Assert.Equal(1, count)); + int nested = 0; + queue.Retire(new Retirement(() => queue.Retire(new Retirement(() => nested++)))); + Assert.Equal(1, queue.Collect()); + Assert.Equal(0, nested); + Assert.Equal(1, queue.Collect()); + Assert.Equal(1, nested); + queue.DisposeAll(); + Assert.Equal(1, nested); + } + + [Fact] + public void RecycledVulkanHandlesDoNotAliasDescriptorCacheEntries() + { + DescriptorSetContents Entry(ulong lifetime) => new(1, 1, + new[] { new SamplerBindingValue(0, new ImageView(123), new Sampler(456), Resource: lifetime) }, + new[] { new BufferBindingValue(1, new Silk.NET.Vulkan.Buffer(789), 0, 256, Resource: lifetime) }); + var cache = new Dictionary(); + cache.Add(Entry(10), "original"); + cache.Add(Entry(11), "replacement"); + Assert.Equal("original", cache[Entry(10)]); + Assert.Equal("replacement", cache[Entry(11)]); + } + + [SkippableFact] + public unsafe void UploadedTexturesAndMeshesSurviveDeletionUntilTheirDrawCompletes() + { + var device = Open(); + try + { + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(location=0) in vec3 position; + void main() { gl_Position = vec4(position, 1); } + """, """ + #version 330 core + uniform sampler2D image; + out vec4 color; + void main() { color = texelFetch(image, ivec2(0), 0); } + """, "retirement-churn"); + device.SetSamplerUnit(program, "image", 0); + int targetTexture = device.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int target = Attach(device, targetTexture, 4, 4); + for (int frame = 0; frame < 48; frame++) + { + device.BeginFrame(); + byte[] expected = { (byte)(frame * 5), 77, 151, 255 }; + int texture; + fixed (byte* pointer = expected) + texture = device.CreateTexture2D(1, 1, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)pointer, false); + int mesh = device.CreateMesh(new MeshData(3, 3) + { + xyz = new[] { -1f, -1f, 0f, 3f, -1f, 0f, -1f, 3f, 0f }, + VerticesCount = 3, Indices = new[] { 0, 1, 2 }, IndicesCount = 3, + mode = EnumDrawMode.Triangles, + }, true); + device.BindFramebuffer(target); + device.SetViewport(0, 0, 4, 4); + device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.UseProgram(program); device.BindTexture(0, texture); + device.DrawMesh(mesh); + device.DeleteMesh(mesh); device.DeleteTexture(texture); + // Deletion precedes submission; readback must still see this frame's upload. + if (frame % 6 == 5) + { + byte[] actual = Read(device, 4, 4); + for (int i = 0; i < actual.Length; i++) Assert.Equal(expected[i % 4], actual[i]); + } + device.Present(); + } + device.DeleteFramebuffer(target); device.DeleteTexture(targetTexture); + GpuTest.AssertClean(device); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + + private const ulong MiB = 1024 * 1024; + private const MemoryPropertyFlags HostMemory = MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit; + + private VulkanContext AllocationContext(List messages) => GpuTest.CreateContext(output, messages); + + [SkippableFact] + public unsafe void PooledAllocationsRemainIsolatedThroughChurnAndReuse() + { + var messages = new List(); + using (var context = AllocationContext(messages)) + { + var live = new List<(VulkanBuffer Buffer, int Stamp)>(); + int serial = 0; + try + { + for (int round = 0; round < 6; round++) + { + while (live.Count < 2048) + { + int stamp = ++serial; + // Different sizes and lifetimes exercise splitting and merging holes. + ulong size = (ulong)(1 + stamp % 16) * 4096; + var buffer = new VulkanBuffer(context, size, BufferUsageFlags.VertexBufferBit, HostMemory); + live.Add((buffer, stamp)); + new Span((void*)buffer.Mapped, (int)size / sizeof(int)).Fill(stamp); + } + foreach (var group in live.GroupBy(entry => entry.Buffer.Allocation.Memory.Handle)) + { + ulong end = 0; + foreach (var entry in group.OrderBy(entry => entry.Buffer.Allocation.Offset)) + { + var allocation = entry.Buffer.Allocation; + Assert.True(allocation.Offset >= end, "Live allocations overlap."); + end = allocation.Offset + allocation.Size; + } + } + foreach (var entry in live) + { + var words = new ReadOnlySpan((void*)entry.Buffer.Mapped, (int)entry.Buffer.Size / sizeof(int)); + Assert.Equal(-1, words.IndexOfAnyExcept(entry.Stamp)); + } + Assert.InRange(context.Allocator.BlockCount, 1, 16); + for (int i = live.Count - 1; i >= 0; i--) + if ((i + round) % 3 != 0) { live[i].Buffer.Dispose(); live.RemoveAt(i); } + } + using var commands = new SetupQueue(context); + using var textures = new TextureManager(context, commands.Uploads); + int image = textures.Create(64, 64, Format.R8G8B8A8Unorm); + Assert.DoesNotContain(live, entry => entry.Buffer.Allocation.Memory.Handle == textures.Get(image)!.Allocation.Memory.Handle); + } + finally { foreach (var entry in live) entry.Buffer.Dispose(); } + // Repeated complete release must reuse blocks instead of growing indefinitely. + int blocks = context.Allocator.BlockCount; + for (int round = 0; round < 3; round++) + { + var batch = new List(); + try + { + for (int i = 0; i < 200; i++) + batch.Add(new VulkanBuffer(context, 128 * 1024, BufferUsageFlags.VertexBufferBit, HostMemory)); + Assert.InRange(context.Allocator.BlockCount, 1, blocks); + } + finally { foreach (var buffer in batch) buffer.Dispose(); } + } + } + ValidationAssert.NoErrors(messages); + } + + [SkippableFact] + public unsafe void DedicatedAllocationsAndEmptyPoolTrimmingRespectTheirBoundaries() + { + var messages = new List(); + using (var context = AllocationContext(messages)) + { + var allocator = context.Allocator; + int baseline = allocator.BlockCount; + var info = new BufferCreateInfo { SType = StructureType.BufferCreateInfo, Size = 4096, + Usage = BufferUsageFlags.VertexBufferBit, SharingMode = SharingMode.Exclusive }; + Assert.Equal(Result.Success, context.Api.CreateBuffer(context.Device, &info, null, out var handle)); + MemoryAllocation dedicated = default; + try + { + var requirements = VulkanAllocator.BufferRequirements(context, handle, out _); + dedicated = allocator.Allocate(requirements, HostMemory, true, "dedicated acceptance", + MemoryPoolClass.DeviceBuffers, true, handle, default); + Assert.True(dedicated.Block!.Dedicated); + Assert.Equal(requirements.Size, dedicated.Block.Size); + Assert.Equal(Result.Success, context.Api.BindBufferMemory(context.Device, handle, dedicated.Memory, dedicated.Offset)); + *(int*)dedicated.Mapped = 7919; + Assert.Equal(7919, *(int*)dedicated.Mapped); + } + finally + { + context.Api.DestroyBuffer(context.Device, handle, null); + if (dedicated.Block != null) allocator.Free(dedicated); + } + Assert.Equal(baseline, allocator.BlockCount); + foreach (var pool in new[] { MemoryPoolClass.DeviceBuffers, MemoryPoolClass.Staging }) + { + ulong threshold = VulkanAllocator.BlockSizeOf(pool) / 4; + using var below = new VulkanBuffer(context, threshold / 2, BufferUsageFlags.TransferSrcBit, HostMemory, pool); + using var at = new VulkanBuffer(context, threshold, BufferUsageFlags.TransferSrcBit, HostMemory, pool); + VulkanAllocator.BufferRequirements(context, below.Handle, out bool driverDedicated); + Assert.Equal(driverDedicated, below.Allocation.Block!.Dedicated); + Assert.True(at.Allocation.Block!.Dedicated); + } + allocator.HeapBudgetOverrideForTests = 1; + allocator.AdvanceFrame(); + Assert.Equal(baseline, allocator.BlockCount); + allocator.HeapBudgetOverrideForTests = null; + var first = new VulkanBuffer(context, MiB, BufferUsageFlags.VertexBufferBit, HostMemory); + Assert.False(first.Allocation.Block!.Dedicated); + var snapshot = allocator.Snapshot(); + context.Api.GetPhysicalDeviceMemoryProperties(context.PhysicalDevice, out var memory); + Assert.Equal((int)memory.MemoryHeapCount, snapshot.HeapUsed.Length); + Assert.Equal((int)memory.MemoryHeapCount, snapshot.HeapBudget.Length); + Assert.True(snapshot.HeapUsed[first.Allocation.Block.HeapIndex] >= first.Allocation.Block.Size); + Assert.True(snapshot.ClassBytes[(int)MemoryPoolClass.DeviceBuffers] >= first.Allocation.Block.Size); + for (int heap = 0; heap < memory.MemoryHeapCount; heap++) + { + Assert.True(snapshot.HeapBudget[heap] > 0); + if (!allocator.BudgetExtension) + Assert.Equal((ulong)(memory.MemoryHeaps[heap].Size * VulkanAllocator.FallbackBudgetShare), snapshot.HeapBudget[heap]); + } + first.Dispose(); + for (int i = 0; i < 60; i++) allocator.AdvanceFrame(); + using (var refill = new VulkanBuffer(context, MiB, BufferUsageFlags.VertexBufferBit, HostMemory)) + Assert.Equal(baseline + 1, allocator.BlockCount); + long freed = allocator.Snapshot().EmptyBlocksFreed; + for (int i = 0; i < VulkanAllocator.EmptyBlockFrames - 1; i++) allocator.AdvanceFrame(); + Assert.Equal(baseline + 1, allocator.BlockCount); + allocator.AdvanceFrame(); + Assert.Equal(baseline, allocator.BlockCount); + Assert.Equal(freed + 1, allocator.Snapshot().EmptyBlocksFreed); + using (var pressured = new VulkanBuffer(context, MiB, BufferUsageFlags.VertexBufferBit, HostMemory)) { } + allocator.HeapBudgetOverrideForTests = 1; + allocator.AdvanceFrame(); + Assert.Equal(baseline, allocator.BlockCount); + } + ValidationAssert.NoErrors(messages); + } + + [SkippableFact] + public unsafe void ReBarBudgetFallsBackToMappedStagingMemory() + { + var messages = new List(); + using (var context = AllocationContext(messages)) + { + var allocator = context.Allocator; + allocator.ReBarCapOverrideForTests = 16 * MiB; + var logged = new List(); + allocator.Log = logged.Add; + long misses = allocator.ReBarMisses, stats = VulkanStats.RebarFallbacks; + var buffers = new List(); + try + { + for (int i = 0; i < 40; i++) + { + var buffer = new VulkanBuffer(context, MiB, BufferUsageFlags.UniformBufferBit, + HostMemory | MemoryPropertyFlags.DeviceLocalBit, MemoryPoolClass.ReBar); + buffers.Add(buffer); + *(int*)buffer.Mapped = i + 1; + Assert.Contains(buffer.Allocation.Block!.Class, new[] { MemoryPoolClass.ReBar, MemoryPoolClass.Staging }); + } + Assert.True(allocator.ReBarUsed <= 16 * MiB); + Assert.True(allocator.ReBarMisses > misses); + Assert.True(VulkanStats.RebarFallbacks - stats >= allocator.ReBarMisses - misses); + Assert.Contains(logged, line => line.Contains("ReBAR miss")); + Assert.Contains(buffers, buffer => buffer.Allocation.Block!.Class == MemoryPoolClass.Staging); + if (allocator.HasMemoryType(uint.MaxValue, HostMemory | MemoryPropertyFlags.DeviceLocalBit, 0)) + Assert.Contains(buffers, buffer => buffer.Allocation.Block!.Class == MemoryPoolClass.ReBar); + for (int i = 0; i < buffers.Count; i++) Assert.Equal(i + 1, *(int*)buffers[i].Mapped); + } + finally { foreach (var buffer in buffers) buffer.Dispose(); } + } + ValidationAssert.NoErrors(messages); + } + + [SkippableTheory] + [InlineData(true)] + [InlineData(false)] + public void MeshAllocationPolicyStillProducesCorrectPixels(bool staticMesh) + { + var device = Open(); + try + { + var allocator = device.ContextForTests.Allocator; + allocator.ReBarCapOverrideForTests = 0; + long misses = allocator.ReBarMisses; + int texture = device.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int target = Attach(device, texture, 8, 8); + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(location = 0) in vec3 position; + void main() { gl_Position = vec4(position, 1); } + """, """ + #version 330 core + out vec4 color; + void main() { color = vec4(1); } + """, "allocation-policy"); + int mesh = device.CreateMesh(new MeshData(4, 6) { + xyz = new[] { -1f, -1f, 0f, 0f, -1f, 0f, 0f, 1f, 0f, -1f, 1f, 0f }, + VerticesCount = 4, Indices = new[] { 0, 1, 2, 0, 2, 3 }, IndicesCount = 6, + }, staticMesh); + foreach (int slot in new[] { MeshManager.BufferXyz, -1 }) + { + var buffer = device.MeshesForTests.BufferOf(mesh, slot)!; + Assert.Equal(MemoryPoolClass.DeviceBuffers, buffer.Allocation.Block!.Class); + var flags = allocator.FlagsOf(buffer.Allocation.Block.TypeIndex); + if (!staticMesh) Assert.NotEqual(IntPtr.Zero, buffer.Mapped); + else + { + Assert.True((flags & MemoryPropertyFlags.DeviceLocalBit) != 0); + var requirements = VulkanAllocator.BufferRequirements(device.ContextForTests, buffer.Handle, out _); + if (allocator.HasMemoryType(requirements.MemoryTypeBits, MemoryPropertyFlags.DeviceLocalBit, MemoryPropertyFlags.HostVisibleBit)) + Assert.Equal(IntPtr.Zero, buffer.Mapped); + } + } + for (int frame = 0; frame < 4; frame++) + { + device.BeginFrame(); device.BindFramebuffer(target); device.SetViewport(0, 0, 8, 8); + device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.UseProgram(program); + device.ClearColor(0, 0, 0, 0, 1); + if (staticMesh) device.DrawMesh(mesh); + else device.DrawMeshMulti(mesh, new[] { 0, 0 }, new[] { 6 }, 1, false); + byte[] pixels = Read(device, 8, 8); + for (int y = 0; y < 8; y++) + for (int x = 0; x < 8; x++) + for (int channel = 0; channel < 4; channel++) + Assert.Equal((byte)(channel == 3 || x < 4 ? 255 : 0), pixels[(y * 8 + x) * 4 + channel]); + device.Present(); + } + if (!staticMesh) Assert.True(allocator.ReBarMisses > misses); + device.DeleteMesh(mesh); device.DeleteFramebuffer(target); device.DeleteTexture(texture); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void YoungDescriptorsReuseSlotStorageAndThenBecomeCached() + { + var device = Open(); + try + { + device.ShortLivedFramesForTests = 4; + int target = Attach(device, device.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), 4, 4); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D sky; + out vec4 color; + void main() { color = texelFetch(sky, ivec2(0), 0); } + """, "descriptor-age"); + device.SetSamplerUnit(program, "sky", 0); + for (int i = 0; i < 5; i++) { device.BeginFrame(); device.Present(); } + byte[] pixel = { 40, 120, 200, 255 }; + int texture; + fixed (byte* pointer = pixel) texture = device.CreateTexture2D(1, 1, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)pointer, false); + int cached = -1; + var poolCounts = new Dictionary(); + for (int frame = 0; frame < 10; frame++) + { + device.BeginFrame(); + int slot = device.CurrentSlotForTests; + var arena = device.DescriptorArenaForTests(slot); + Assert.Equal(0, arena.SetsThisFrame); + device.BindTexture(0, texture); + Draw(device, target, program, 4, 4); + long allocations = arena.Allocations; + for (int i = 0; i < 16; i++) device.DrawFullscreenTriangle(); + Assert.Equal(allocations, arena.Allocations); + if (frame == 0) cached = device.CachedDescriptorSets; + if (frame < 3) { Assert.Equal(1, arena.SetsThisFrame); Assert.Equal(cached, device.CachedDescriptorSets); } + if (frame >= 4) { Assert.Equal(0, arena.SetsThisFrame); Assert.Equal(cached + 1, device.CachedDescriptorSets); } + if (poolCounts.TryGetValue(slot, out int pools)) Assert.Equal(pools, arena.PoolCount); + else poolCounts.Add(slot, arena.PoolCount); + byte[] actual = Read(device, 4, 4); + for (int i = 0; i < actual.Length; i++) Assert.Equal(pixel[i % 4], actual[i]); + device.Present(); + } + device.DeleteTexture(texture); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } +} diff --git a/Optimum.Render.Vulkan.Tests/ShaderBindingTests.cs b/Optimum.Render.Vulkan.Tests/ShaderBindingTests.cs new file mode 100644 index 00000000..c5c0d291 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ShaderBindingTests.cs @@ -0,0 +1,329 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class ShaderBindingTests(ITestOutputHelper output) +{ + private const string Triangle = """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); } + """; + private sealed class Clock : ITimelineClock + { + public ulong FrameRecorded { get; set; } + public ulong FrameCompleted { get; set; } + public ulong TransferRecorded { get; set; } + public ulong TransferCompleted { get; set; } + } + + [Fact] + public void BindlessSlotsRemainReservedUntilTheirRetirementCompletes() + { + var clock = new Clock { FrameRecorded = 7, FrameCompleted = 5 }; + var book = new BindlessSlotBook(clock, Enumerable.Repeat(3u, BindlessKinds.Count).ToArray()); + BindlessSlotKey Key(ulong id) => new(id, TextureKind.Texture2D, SamplerState.Default, ImageLayout.ShaderReadOnlyOptimal); + uint first = book.Acquire(Key(1), out bool created); + Assert.True(created); Assert.NotEqual(0u, first); + Assert.Equal(first, book.Acquire(Key(1), out created)); Assert.False(created); + Assert.Equal(1, book.Release(1)); + uint second = book.Acquire(Key(2), out _); + Assert.NotEqual(first, second); Assert.NotEqual(0u, second); + Assert.Equal(0u, book.Acquire(Key(3), out _)); Assert.Equal(1, book.Exhausted); + var freed = new List<(TextureKind, uint)>(); + clock.FrameCompleted = 6; Assert.Equal(0, book.Collect(freed)); + clock.FrameCompleted = 7; Assert.Equal(1, book.Collect(freed)); + Assert.Equal(new[] { (TextureKind.Texture2D, first) }, freed); + Assert.Equal(first, book.Acquire(Key(3), out _)); + } + + [Fact] + public void SamplerVariantsAreDistinctAndEvictTheLeastRecentlyUsedVariant() + { + var book = new BindlessSlotBook(new Clock { FrameRecorded = 1 }, Enumerable.Repeat(32u, BindlessKinds.Count).ToArray()); + BindlessSlotKey Key(int bias) => new(9, TextureKind.Texture2D, SamplerState.Default with { LodBias = bias }, ImageLayout.ShaderReadOnlyOptimal); + uint first = book.Acquire(Key(0), out _); + var slots = new HashSet { first }; + for (int i = 1; i < BindlessSlotBook.MaxVariantsPerTexture; i++) Assert.True(slots.Add(book.Acquire(Key(i), out _))); + Assert.Equal(first, book.Acquire(Key(0), out _)); + book.Acquire(Key(BindlessSlotBook.MaxVariantsPerTexture), out _); + Assert.Equal(1, book.PendingRetirements); + Assert.Equal(first, book.Acquire(Key(0), out bool created)); Assert.False(created); + book.Acquire(Key(1), out created); Assert.True(created); + } + + private VulkanDevice Open(bool poison = false) => GpuTest.CreateDevice(output, device => { + var configure = device.ConfigureContextOptions; + device.ConfigureContextOptions = options => { configure?.Invoke(options); options.Poison = poison; }; + }); + + private static unsafe int Texture(VulkanDevice device, byte r, byte g, byte b) + { + byte[] pixel = { r, g, b, 255 }; + fixed (byte* pointer = pixel) return device.CreateTexture2DRaw(1, 1, 0x8058, (IntPtr)pointer, 4); + } + + private static (int Image, int Target) Target(VulkanDevice device) + { + int image = device.CreateTexture2DRaw(4, 4, 0x8058, IntPtr.Zero, 4), target = device.CreateFramebuffer(4, 4); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + return (image, target); + } + + private static void Draw(VulkanDevice device, int program, int target) + { + device.BindFramebuffer(target); device.UseProgram(program); device.SetViewport(0, 0, 4, 4); + device.SetDepthTest(false); device.SetDepthMask(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.DrawFullscreenTriangle(); + } + + private static void Pixels(VulkanDevice device, int image, byte red, byte green, byte blue) + { + byte[] pixels = device.ReadBackLevel0ForTests(image); + Assert.Equal(4 * 4 * 4, pixels.Length); + byte[] expected = { red, green, blue, 255 }; + for (int i = 0; i < pixels.Length; i++) Assert.InRange((int)pixels[i], Math.Max(0, expected[i % 4] - 1), Math.Min(255, expected[i % 4] + 1)); + } + + [SkippableTheory] + [InlineData(false)] + [InlineData(true)] + public void TextureReplacementAndWrongTypesUseTheCorrectBindlessSlots(bool poison) + { + var device = Open(poison); + try + { + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D image; + out vec4 color; + void main() { color = texture(image, vec2(0.5)); } + """, "binding-lifetime"); + device.SetSamplerUnit(program, "image", 0); + var first = Target(device); var replacement = Target(device); var placeholder = Target(device); + var table = device.BindlessForTests; + int red = Texture(device, 255, 0, 0); + device.BeginFrame(); + uint retiredSlot = table.Resolve(red, TextureKind.Texture2D, SamplerState.Default); + device.BindTexture(0, red); Draw(device, program, first.Target); + device.DeleteTexture(red); + int green = Texture(device, 0, 255, 0); + Assert.Equal(red, green); + uint newSlot = table.Resolve(green, TextureKind.Texture2D, SamplerState.Default); + Assert.NotEqual(retiredSlot, newSlot); Assert.NotEqual(0u, newSlot); + device.BindTexture(0, green); Draw(device, program, replacement.Target); + device.BindTexture(0, 0); Draw(device, program, placeholder.Target); + device.Present(); + for (int i = 0; i < 4; i++) { device.BeginFrame(); device.Present(); } + Assert.Equal(0, table.PendingRetirements); + device.BeginFrame(); + Pixels(device, first.Image, 255, 0, 0); Pixels(device, replacement.Image, 0, 255, 0); + Pixels(device, placeholder.Image, poison ? (byte)255 : (byte)0, 0, poison ? (byte)255 : (byte)0); + foreach (var wrong in new[] { TextureKind.Shadow2D, TextureKind.Texture2DArray, TextureKind.TextureCube, + TextureKind.Texture3D, TextureKind.SignedTexture2D, TextureKind.UnsignedTexture2D }) + Assert.Equal(0u, table.Resolve(green, wrong, SamplerState.Default)); + int blue = Texture(device, 0, 0, 255); + Assert.Equal(retiredSlot, table.Resolve(blue, TextureKind.Texture2D, SamplerState.Default)); + device.BindTexture(0, blue); Draw(device, program, first.Target); Pixels(device, first.Image, 0, 0, 255); + var held = device.TexturesForTests.Get(blue)!; + device.DeleteTexture(blue); + Assert.Equal(0u, table.Resolve(held, TextureKind.Texture2D, SamplerState.Default)); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void SwitchingProgramsKeepsSamplerAssignmentsAndDrawSnapshots() + { + var device = Open(); + try + { + int single = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D colour; + out vec4 color; + void main() { color = texture(colour, vec2(0.5)); } + """, "single-sampler"); + int pair = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D first; + uniform sampler2D second; + uniform float weight; + out vec4 color; + void main() { color = mix(texture(first, vec2(0.5)), texture(second, vec2(0.5)), weight); } + """, "paired-samplers"); + device.SetSamplerUnit(single, "colour", 0); + device.SetSamplerUnit(pair, "first", 1); device.SetSamplerUnit(pair, "second", 2); + device.SetUniform(pair, device.GetUniformLocation(pair, "weight"), 1f); + int red = Texture(device, 255, 0, 0), green = Texture(device, 0, 255, 0), blue = Texture(device, 0, 0, 255); + var targets = Enumerable.Range(0, 4).Select(_ => Target(device)).ToArray(); + device.BeginFrame(); + var textureBinds = device.TextureSetBindsForTests; + var frameBinds = device.FrameSetBindsForTests; + device.BindTexture(0, red); device.BindTexture(1, green); device.BindTexture(2, blue); + Draw(device, single, targets[0].Target); Draw(device, pair, targets[1].Target); + device.BindTexture(0, green); Draw(device, single, targets[2].Target); + device.BindTexture(2, red); Draw(device, pair, targets[3].Target); + Assert.InRange(device.TextureSetBindsForTests - textureBinds, 0, 1); + Assert.Equal(frameBinds, device.FrameSetBindsForTests); + Pixels(device, targets[0].Image, 255, 0, 0); Pixels(device, targets[1].Image, 0, 0, 255); + Pixels(device, targets[2].Image, 0, 255, 0); Pixels(device, targets[3].Image, 255, 0, 0); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void ShadowSamplingUsesComparisonAndTheFarPlanePlaceholder() + { + var device = Open(); + try + { + float depth = 0.5f; + int texture = device.CreateTexture2DRaw(1, 1, 0x8CAC, (IntPtr)(&depth), 4); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2DShadow probeDepth; + uniform float referenceDepth; + out vec4 color; + void main() { color = vec4(texture(probeDepth, vec3(0.5, 0.5, referenceDepth)), 0, 0, 1); } + """, "shadow-slot"); + device.SetSamplerUnit(program, "probeDepth", 0); + int reference = device.GetUniformLocation(program, "referenceDepth"); + var targets = Enumerable.Range(0, 3).Select(_ => Target(device)).ToArray(); + device.BeginFrame(); + for (int i = 0; i < 3; i++) + { + device.BindTexture(0, i == 2 ? 0 : texture); + device.SetUniform(program, reference, i == 0 ? 0.25f : 0.75f); + Draw(device, program, targets[i].Target); + } + for (int i = 0; i < 3; i++) Pixels(device, targets[i].Image, i == 1 ? (byte)0 : (byte)255, 0, 0); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void FrameGlobalsAreSharedOnlyByProgramsWithTheirOwningInclude() + { + var device = Open(); + try + { + int Link(bool shared) + { + var vertex = new Shader(EnumShaderType.VertexShader, Triangle, "globals.vsh"); + var fragment = new Shader(EnumShaderType.FragmentShader, """ + #version 330 core + uniform float zNear; + uniform float tint; + out vec4 color; + void main() { color = vec4(zNear, tint, 0, 1); } + """, "globals.fsh"); + Assert.True(device.CompileShader(vertex)); Assert.True(device.CompileShader(fragment)); + var program = new ShaderProgram { PassName = "binding-globals", VertexShader = vertex, FragmentShader = fragment }; + if (shared) program.includes.Add("fogandlight.fsh"); + int id = device.LinkProgram(program); Assert.True(id > 0, device.GetError()); return id; + } + int writer = Link(true), reader = Link(true), isolated = Link(false); + device.SetUniform(writer, device.GetUniformLocation(writer, "zNear"), 0.25f); + device.SetUniform(reader, device.GetUniformLocation(reader, "tint"), 0.75f); + device.SetUniform(isolated, device.GetUniformLocation(isolated, "zNear"), 0.5f); + device.SetUniform(isolated, device.GetUniformLocation(isolated, "tint"), 0.25f); + var targets = Enumerable.Range(0, 3).Select(_ => Target(device)).ToArray(); + device.BeginFrame(); + Draw(device, reader, targets[0].Target); Draw(device, isolated, targets[1].Target); + device.SetUniform(writer, device.GetUniformLocation(writer, "zNear"), 0.75f); + Draw(device, reader, targets[2].Target); + Pixels(device, targets[0].Image, 64, 191, 0); Pixels(device, targets[1].Image, 128, 64, 0); + Pixels(device, targets[2].Image, 191, 191, 0); device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void AnimationStorageKeepsEachDrawsSnapshot() + { + var device = Open(); + try + { + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(std140) uniform Animation { mat4 values[4]; } bones; + out vec4 tint; + void main() { + gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); + tint = bones.values[2][3]; + } + """, """ + #version 330 core + in vec4 tint; + out vec4 color; + void main() { color = tint; } + """, "animation-snapshots"); + int block = device.CreateUniformBuffer(program, 0, "Animation", 256); + var targets = Enumerable.Range(0, 4).Select(_ => Target(device)).ToArray(); + device.BeginFrame(); + for (int i = 0; i < targets.Length; i++) + { + var values = new float[64]; values[44] = (20 + i * 60) / 255f; values[45] = 1; values[47] = 1; + fixed (float* pointer = values) device.UpdateUniformBuffer(block, (IntPtr)pointer, 0, 256); + Draw(device, program, targets[i].Target); + } + for (int i = 0; i < targets.Length; i++) Pixels(device, targets[i].Image, (byte)(20 + i * 60), 255, 0); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void RetainedFrameTextureIsReadableAfterBeingAnAttachment() + { + var device = Open(); + try + { + float initial = 0.25f; + int depth = device.CreateTexture2DRaw(1, 1, 0x8CAC, (IntPtr)(&initial), 4); + var target = Target(device); + int depthTarget = device.CreateFramebuffer(1, 1); + device.AttachTexture(depthTarget, EnumFramebufferAttachment.DepthAttachment, depth, 0); + int reader = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D liquidDepth; + out vec4 color; + void main() { color = vec4(texture(liquidDepth, vec2(0.5)).r, 0, 0, 1); } + """, "retained-frame-texture"); + int writer = GpuTest.LinkProgram(device, Triangle, "#version 330 core\nvoid main() {}", "depth-only"); + device.SetSamplerUnit(reader, "liquidDepth", 0); + device.BeginFrame(); device.BindTexture(0, depth); Draw(device, reader, target.Target); + device.BindTexture(0, 0); device.BindFramebuffer(depthTarget); device.UseProgram(writer); + device.SetViewport(0, 0, 1, 1); device.SetDepthTest(true); device.SetDepthMask(true); device.SetDepthFunc(0x207); + device.DrawFullscreenTriangle(); + var pipeline = device.RequestNativePipeline(new NativePipelineDescription { + ProgramId = reader, Blend = new[] { AttachmentBlend.For(false, EnumBlendMode.Standard) }, + Targets = device.NativeTargetFormats(target.Target, 1)!, + }, out string error); + Assert.NotNull(pipeline); + Assert.True(device.BeginNativePass(new NativePassDescription { Name = "retained-frame-read", FramebufferId = target.Target })); + Assert.True(device.DrawNativeFullscreen(pipeline!, ReadOnlySpan.Empty), error); + device.EndNativePass(); Pixels(device, target.Image, 128, 0, 0); device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } +} diff --git a/Optimum.Render.Vulkan.Tests/ShaderCorpus.cs b/Optimum.Render.Vulkan.Tests/ShaderCorpus.cs new file mode 100644 index 00000000..3c6c7094 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ShaderCorpus.cs @@ -0,0 +1,679 @@ +// Source: Optimum.Render.Vulkan.Tests/ShaderCorpus.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Text.RegularExpressions; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; + +/// +/// Loads the game's shaders the way the client does, so translation is tested +/// against what actually reaches the driver rather than against the files on disk. +/// +/// Two steps matter. ShaderRegistry expands #include before compiling, and +/// it de-duplicates per program, so a shader that pulls in fogandlight.fsh twice +/// gets it once. And registerDefaultShaderCodePrefixes prepends a block of +/// #defines whose values change which declarations survive the +/// preprocessor - a shader compiled with SSAOLEVEL 0 declares different uniforms +/// than the same file at SSAOLEVEL 2. +/// +internal static class ShaderCorpus +{ + /// + /// The vanilla assets, or null when this checkout has not bootstrapped. The + /// shaders are proprietary and never committed, so tests that need them skip + /// rather than fail when they are absent. + /// + public static string? AssetRoot => _assetRoot ??= FindAssetRoot(); + private static string? _assetRoot; + + private static string? FindAssetRoot() + { + string? fromEnvironment = Environment.GetEnvironmentVariable("VINTAGE_STORY_ASSETS"); + if (!string.IsNullOrEmpty(fromEnvironment) && + Directory.Exists(Path.Combine(fromEnvironment, "shaders"))) + { + return fromEnvironment; + } + + DirectoryInfo? directory = new(AppContext.BaseDirectory); + while (directory != null) + { + string candidate = Path.Combine( + directory.FullName, ".vanilla", "win-x64", "vintagestory", "assets", "game"); + if (Directory.Exists(Path.Combine(candidate, "shaders"))) + { + return candidate; + } + directory = directory.Parent; + } + + return null; + } + + /// The repository root, found by walking up to the solution file. + public static string RepositoryRoot => _repositoryRoot ??= FindRepositoryRoot(); + private static string? _repositoryRoot; + + private static string FindRepositoryRoot() + { + DirectoryInfo? directory = new(AppContext.BaseDirectory); + while (directory != null) + { + if (File.Exists(Path.Combine(directory.FullName, "VintageStory.slnx"))) + { + return directory.FullName; + } + directory = directory.Parent; + } + throw new DirectoryNotFoundException("Repository root with VintageStory.slnx not found."); + } + + /// + /// Every shader file, with Optimum's own overlays replacing their vanilla + /// counterparts - which is what `make deploy` copies over the install, so it + /// is what actually runs. + /// + public static Dictionary LoadShaderFiles() + { + var files = new Dictionary(StringComparer.OrdinalIgnoreCase); + + if (AssetRoot != null) + { + foreach (string path in Directory.EnumerateFiles(Path.Combine(AssetRoot, "shaders"))) + { + files[Path.GetFileName(path)] = File.ReadAllText(path); + } + } + + string overlays = Path.Combine(RepositoryRoot, "sources", "shaders"); + if (Directory.Exists(overlays)) + { + foreach (string path in Directory.EnumerateFiles(overlays)) + { + files[Path.GetFileName(path)] = File.ReadAllText(path); + } + } + + return files; + } + + /// + /// The shader includes, with Optimum's own overlays replacing their vanilla + /// counterparts - the same relationship has, + /// and the same one `make deploy` and the package scripts produce on disk. + /// Without the overlay the corpus would translate the vanilla vertexwarp.vsh + /// while the client runs Optimum's WarpState one. + /// + public static Dictionary LoadIncludes() + { + var includes = new Dictionary(StringComparer.OrdinalIgnoreCase); + if (AssetRoot != null) + { + string directory = Path.Combine(AssetRoot, "shaderincludes"); + if (Directory.Exists(directory)) + { + foreach (string path in Directory.EnumerateFiles(directory)) + { + includes[Path.GetFileName(path)] = File.ReadAllText(path); + } + } + } + + string overlays = Path.Combine(RepositoryRoot, "sources", "shaderincludes"); + if (Directory.Exists(overlays)) + { + foreach (string path in Directory.EnumerateFiles(overlays)) + { + includes[Path.GetFileName(path)] = File.ReadAllText(path); + } + } + + return includes; + } + + /// + /// Program base names that have both a vertex and a fragment shader, which is + /// what ShaderRegistry requires of a loadable program. + /// + public static List ProgramNames(Dictionary files) + { + return files.Keys + .Where(name => name.EndsWith(".vsh", StringComparison.OrdinalIgnoreCase)) + .Select(name => name[..^4]) + .Where(name => files.ContainsKey(name + ".fsh")) + .OrderBy(name => name, StringComparer.Ordinal) + .ToList(); + } + + private static readonly Regex IncludePattern = + new(@"^#include\s+(.*)", RegexOptions.Multiline | RegexOptions.Compiled); + + /// + /// Mirrors ShaderRegistry.HandleIncludes, including its de-duplication: a + /// file already pulled into this program expands to nothing the second time. + /// + public static string ExpandIncludes( + string code, Dictionary includes, HashSet? seen = null) + { + seen ??= new HashSet(StringComparer.OrdinalIgnoreCase); + + return IncludePattern.Replace(code, match => + { + string filename = match.Groups[1].Value.Trim().ToLowerInvariant(); + if (!seen.Add(filename)) return ""; + return includes.TryGetValue(filename, out string? included) + ? ExpandIncludes(included, includes, seen) + : ""; + }); + } + + /// A set of the defines the client injects per program. + public sealed class ShaderVariant + { + public string Name = ""; + public int Fxaa; + public int SsaoLevel; + public int Bloom; + public int GodRays; + public int ShadowQuality; + public int DynLights; + public int UseSsbo; + public int WavingStuff = 1; + public int FoamEffect = 1; + public int ShinyEffect = 1; + public int NormalView; + public int UseOit = 1; + public int GreedyMesh; + public float MinBright; + public int MaxAnimatedElements = 35; + /// TAA motion-vector writers compiled in (OptimumConfig.EffectiveTaa). + public int TaaMotion; + /// Primary colour attachment the motion texture occupies: 4 with the SSAO G-buffer, 2 without. + public int TaaMotionLocation = 2; + /// ShaderRegistry's OPTIMUMAO: 1 while the Vulkan platform runs GTAO. + public int OptimumAo; + + /// + /// Defines a caller put on the program itself before the engine's block, + /// the way ModSystemFpHands stamps ALLOWDEPTHOFFSET on its two private + /// copies of standard and entityanimated. Those copies are the only place + /// the gl_FragDepth writer exists, so without this the corpus never + /// translates it. + /// + public string ExtraPrefix = ""; + + public override string ToString() => Name; + } + + /// + /// Variants chosen to move every define that gates a declaration. The two + /// extremes catch the common cases; the middle rows catch the combinations + /// where one feature is on and its neighbours are not, which is where a + /// conditional uniform is most likely to be missed. + /// + public static IEnumerable Variants() + { + // USEOIT tracks ShaderProgramBase.Oit, which defaults to true for every + // program; only the second Entityanimated registration turns it off. The + // shaders that call into oit.fsh do so unconditionally, so a global + // USEOIT 0 is not a configuration the client can produce. + yield return new ShaderVariant + { + Name = "everything-off", + WavingStuff = 0, FoamEffect = 0, ShinyEffect = 0, + }; + yield return new ShaderVariant + { + Name = "everything-on", + Fxaa = 1, SsaoLevel = 2, Bloom = 1, GodRays = 2, ShadowQuality = 2, + DynLights = 8, UseSsbo = 1, NormalView = 1, GreedyMesh = 1, MinBright = 0.1f, + }; + yield return new ShaderVariant + { + Name = "ssao-only", + SsaoLevel = 1, DynLights = 4, + }; + yield return new ShaderVariant + { + Name = "shadows-and-ssbo", + ShadowQuality = 2, DynLights = 8, UseSsbo = 1, + }; + // TAA on, without the SSAO G-buffer (motion at attachment 2) and with it + // (attachment 4): the motion-vector writers only exist in these, and the + // two rows move the output location the same way the client does. + yield return new ShaderVariant + { + Name = "taa-no-ssao", + TaaMotion = 1, TaaMotionLocation = 2, + }; + yield return new ShaderVariant + { + Name = "taa-with-ssao", + SsaoLevel = 2, DynLights = 4, TaaMotion = 1, TaaMotionLocation = 4, + }; + // TAA on with the waving-stuff, foam and shiny settings off. Those are + // ordinary client settings, and WAVINGSTUFF in particular is what gates + // the bodies of every vertexwarp function the motion writers replay for + // the previous frame - so with TAA on it is a shipped combination that + // no other row produced (the "everything-off" row carries TAAMOTION 0). + // The Vulkan AO row: TAA with the SSAO G-buffer and OPTIMUMAO 1, the combination that + // compiles in the class-channel writes (chunkopaque's no-cull flag, the hand-view class + // in standard and entityanimated) and scene-ssao's GTAO compose branch. + yield return new ShaderVariant + { + Name = "taa-with-gtao", + SsaoLevel = 2, DynLights = 4, TaaMotion = 1, TaaMotionLocation = 4, OptimumAo = 1, + }; + yield return new ShaderVariant + { + Name = "taa-no-waving", + TaaMotion = 1, TaaMotionLocation = 2, + WavingStuff = 0, FoamEffect = 0, ShinyEffect = 0, + }; + } + + /// + /// Reproduces registerDefaultShaderCodePrefixes, including Optimum's own + /// greedy-mesh defines from the ShaderRegistry patch. + /// + public static string PrefixFor(EnumShaderType stage, ShaderVariant variant) + { + var lines = new List(); + + if (stage == EnumShaderType.FragmentShader) + { + lines.Add($"#define FXAA {variant.Fxaa}"); + lines.Add($"#define SSAOLEVEL {variant.SsaoLevel}"); + lines.Add($"#define NORMALVIEW {variant.NormalView}"); + lines.Add($"#define BLOOM {variant.Bloom}"); + lines.Add($"#define GODRAYS {variant.GodRays}"); + lines.Add($"#define FOAMEFFECT {variant.FoamEffect}"); + lines.Add($"#define SHINYEFFECT {variant.ShinyEffect}"); + lines.Add($"#define SHADOWQUALITY {variant.ShadowQuality}"); + lines.Add($"#define DYNLIGHTS {variant.DynLights}"); + lines.Add($"#define USEOIT {variant.UseOit}"); + lines.Add($"#define GREEDYMESH {variant.GreedyMesh}"); + lines.Add($"#define GREEDYMESH_GRAD 0"); + lines.Add($"#define TAAMOTION {variant.TaaMotion}"); + lines.Add($"#define TAAMOTIONLOCATION {variant.TaaMotionLocation}"); + lines.Add($"#define OPTIMUMAO {variant.OptimumAo}"); + } + else + { + lines.Add($"#define USESSBO {variant.UseSsbo}"); + lines.Add($"#define WAVINGSTUFF {variant.WavingStuff}"); + lines.Add($"#define FOAMEFFECT {variant.FoamEffect}"); + lines.Add($"#define SSAOLEVEL {variant.SsaoLevel}"); + lines.Add($"#define NORMALVIEW {variant.NormalView}"); + lines.Add($"#define SHINYEFFECT {variant.ShinyEffect}"); + lines.Add($"#define GODRAYS {variant.GodRays}"); + lines.Add($"#define MINBRIGHT {variant.MinBright.ToString(System.Globalization.CultureInfo.InvariantCulture)}"); + lines.Add($"#define SHADOWQUALITY {variant.ShadowQuality}"); + lines.Add($"#define DYNLIGHTS {variant.DynLights}"); + lines.Add($"#define MAXANIMATEDELEMENTS {variant.MaxAnimatedElements}"); + lines.Add($"#define GREEDYMESH {variant.GreedyMesh}"); + lines.Add($"#define TAAMOTION {variant.TaaMotion}"); + lines.Add($"#define TAAMOTIONLOCATION {variant.TaaMotionLocation}"); + lines.Add($"#define OPTIMUMAO {variant.OptimumAo}"); + } + + string prefix = string.Join("\r\n", lines) + "\r\n"; + // The client puts the program's own PrefixCode first and appends the + // engine block to it (ShaderRegistry.registerDefaultShaderCodePrefixes + // does `PrefixCode = PrefixCode + ...`), so a caller's defines lead. + return variant.ExtraPrefix.Length > 0 + ? variant.ExtraPrefix + "\r\n" + prefix + : prefix; + } + + /// Builds the two stages of one program, ready for translation. + public static List BuildProgram( + string programName, + Dictionary files, + Dictionary includes, + ShaderVariant variant) + { + var stages = new List(); + + foreach ((string extension, EnumShaderType stage) in new[] + { + (".vsh", EnumShaderType.VertexShader), + (".fsh", EnumShaderType.FragmentShader), + (".gsh", EnumShaderType.GeometryShader), + }) + { + if (!files.TryGetValue(programName + extension, out string? code)) continue; + + stages.Add(new ShaderStageSource + { + Stage = stage, + Code = ExpandIncludes(code, includes), + PrefixCode = PrefixFor(stage, variant), + Filename = programName + extension, + }); + } + + return stages; + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/NativeShaderTree.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Text.RegularExpressions; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; + +/// +/// The native shader tree (sources/shaders-vk) as the tests see it: the include files, the +/// machine-readable header every ported include carries, include closures, and a shaderc compiler +/// that resolves the includes. +/// +/// A port's header states what the GLSL 330 file declared and where each name now comes from: +/// +/// // optimum-port-of: fogandlight.fsh +/// // optimum-port: verbatim | transformed +/// // optimum-frame-owner: fogandlight.fsh (members this file owns read the frame block) +/// // optimum-frame-texture: sampler2DShadow shadowMapFar +/// // optimum-program-uniform: float flatFogDensity (declared by the including program) +/// // optimum-program-symbol: vec4 rgbaFog (a non-uniform name the program supplies) +/// +/// +internal static class NativeShaderTree +{ + public static string IncludeDirectory => + Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk", "include"); + + public static string Read(string include) => + File.ReadAllText(Path.Combine(IncludeDirectory, include)).Replace("\r\n", "\n"); + + public static List IncludeNames() => + Directory.EnumerateFiles(IncludeDirectory, "*.glsl") + .Select(path => Path.GetFileName(path)!) + .OrderBy(name => name, StringComparer.Ordinal) + .ToList(); + + public readonly record struct Declaration(string Type, string Name, string Text); + + public sealed class Port + { + public string Include = ""; + public string GameFile = ""; + public string Kind = ""; + public string? FrameOwner; + public readonly List FrameTextures = new(); + public readonly List ProgramUniforms = new(); + public readonly List ProgramSymbols = new(); + } + + private static readonly Regex HeaderLine = new(@"^// optimum-([a-z-]+): (.+)$", RegexOptions.Multiline); + private static readonly Regex DeclarationText = new(@"^(\w+)\s+(\w+)\s*(\[[^\]]*\])?\s*(=.*)?$"); + private static readonly Regex IncludeLine = new(@"^\s*#include\s+""([^""]+)""", RegexOptions.Multiline); + + /// The header of a ported include, or null for a file that ports nothing (bindings, motion, ...). + public static Port? PortOf(string include) + { + string text = Read(include); + var port = new Port { Include = include }; + foreach (Match line in HeaderLine.Matches(text)) + { + string value = line.Groups[2].Value.Trim(); + switch (line.Groups[1].Value) + { + case "port-of": port.GameFile = value; break; + case "port": port.Kind = value; break; + case "frame-owner": port.FrameOwner = value; break; + case "frame-texture": port.FrameTextures.Add(Parse(value)); break; + case "program-uniform": port.ProgramUniforms.Add(Parse(value)); break; + case "program-symbol": port.ProgramSymbols.Add(Parse(value)); break; + } + } + return port.GameFile.Length == 0 ? null : port; + } + + private static Declaration Parse(string value) + { + Match match = DeclarationText.Match(value); + if (!match.Success) throw new FormatException("bad header declaration: " + value); + return new Declaration(match.Groups[1].Value, match.Groups[2].Value, + match.Groups[1].Value + " " + match.Groups[2].Value + match.Groups[3].Value); + } + + public static IEnumerable DirectIncludes(string include) => + IncludeLine.Matches(Read(include)).Select(match => match.Groups[1].Value); + + /// The include and everything it pulls in, transitively. + public static List Closure(string include) + { + var seen = new List(); + var pending = new Stack(); + pending.Push(include); + while (pending.Count > 0) + { + string next = pending.Pop(); + if (seen.Contains(next)) continue; + seen.Add(next); + foreach (string child in DirectIncludes(next)) pending.Push(child); + } + return seen; + } + + /// The stages an include can be compiled in: a .vsh port is vertex-only, .fsh and motion fragment-only. + public static EnumShaderType[] StagesOf(string include) + { + Port? port = PortOf(include); + string gameFile = port?.GameFile ?? ""; + if (gameFile.EndsWith(".vsh", StringComparison.Ordinal)) return new[] { EnumShaderType.VertexShader }; + if (gameFile.EndsWith(".fsh", StringComparison.Ordinal) || include == "motion.glsl") + { + return new[] { EnumShaderType.FragmentShader }; + } + return new[] { EnumShaderType.VertexShader, EnumShaderType.FragmentShader }; + } + + public static bool TryCreateCompiler(out ShaderCompiler? compiler, out string reason) + { + try + { + compiler = new ShaderCompiler(); + reason = ""; + return true; + } + catch (Exception error) when (error is DllNotFoundException or InvalidOperationException) + { + compiler = null; + reason = "shaderc unavailable: " + error.Message; + return false; + } + } + + public static ShaderCompileResult Compile(ShaderCompiler compiler, string source, EnumShaderType stage, string name) + { + string extension = stage == EnumShaderType.VertexShader ? ".vert" : ".frag"; + return compiler.CompileNative(source, Path.Combine(IncludeDirectory, name + extension), stage, IncludeDirectory); + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/SpirvReader.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using System.Text; + +/// +/// The few SPIR-V facts the native-shader tests check - names, decorations, struct members, +/// variables and specialization constants - read straight from the word stream. No reflection +/// library exists in the tree (docs/vulkan.md plans +/// Shaders/SpirvReflection.cs); until it does, this is deliberately minimal. +/// +internal sealed class SpirvReader +{ + private const uint Magic = 0x07230203; + + private const ushort OpName = 5; + private const ushort OpMemberName = 6; + private const ushort OpTypeInt = 21; + private const ushort OpTypeFloat = 22; + private const ushort OpTypeArray = 28; + private const ushort OpTypeStruct = 30; + private const ushort OpTypePointer = 32; + private const ushort OpSpecConstant = 50; + private const ushort OpVariable = 59; + private const ushort OpDecorate = 71; + private const ushort OpMemberDecorate = 72; + + public const uint DecorationSpecId = 1; + public const uint DecorationArrayStride = 6; + public const uint DecorationBinding = 33; + public const uint DecorationDescriptorSet = 34; + public const uint DecorationOffset = 35; + + public const uint StorageUniform = 2; + public const uint StoragePushConstant = 9; + + public readonly Dictionary Names = new(); + public readonly Dictionary<(uint Type, uint Member), string> MemberNames = new(); + public readonly Dictionary> Decorations = new(); + public readonly Dictionary<(uint Type, uint Member), Dictionary> MemberDecorations = new(); + public readonly Dictionary Structs = new(); + public readonly Dictionary ArrayElements = new(); + public readonly Dictionary Pointers = new(); + public readonly Dictionary Variables = new(); + public readonly Dictionary ScalarTypes = new(); + public readonly Dictionary SpecConstantTypes = new(); + + public static SpirvReader Parse(byte[] bytes) + { + if (bytes.Length < 20 || bytes.Length % 4 != 0) throw new ArgumentException("not a SPIR-V module"); + var words = new uint[bytes.Length / 4]; + Buffer.BlockCopy(bytes, 0, words, 0, bytes.Length); + if (words[0] != Magic) throw new ArgumentException("bad SPIR-V magic"); + + var reader = new SpirvReader(); + for (int at = 5; at < words.Length;) + { + ushort opcode = (ushort)(words[at] & 0xFFFF); + int count = (int)(words[at] >> 16); + if (count == 0) throw new ArgumentException("zero-length instruction at word " + at); + ReadOnlySpan operands = new ReadOnlySpan(words, at + 1, count - 1); + reader.Take(opcode, operands); + at += count; + } + return reader; + } + + private void Take(ushort opcode, ReadOnlySpan o) + { + switch (opcode) + { + case OpName: Names[o[0]] = ReadString(o[1..]); break; + case OpMemberName: MemberNames[(o[0], o[1])] = ReadString(o[2..]); break; + case OpTypeInt: ScalarTypes[o[0]] = o[2] == 1 ? "int" : "uint"; break; + case OpTypeFloat: ScalarTypes[o[0]] = "float"; break; + case OpTypeArray: ArrayElements[o[0]] = o[1]; break; + case OpTypeStruct: Structs[o[0]] = o[1..].ToArray(); break; + case OpTypePointer: Pointers[o[0]] = (o[1], o[2]); break; + case OpSpecConstant: SpecConstantTypes[o[1]] = o[0]; break; + case OpVariable: Variables[o[1]] = (o[0], o[2]); break; + case OpDecorate: + Get(Decorations, o[0])[o[1]] = o.Length > 2 ? o[2] : 1; + break; + case OpMemberDecorate: + if (!MemberDecorations.TryGetValue((o[0], o[1]), out Dictionary? member)) + { + MemberDecorations[(o[0], o[1])] = member = new Dictionary(); + } + member[o[2]] = o.Length > 3 ? o[3] : 1; + break; + } + } + + /// The block type of the variable at /, or null. + public uint? BlockAt(uint set, uint binding) + { + foreach ((uint id, (uint pointer, uint _)) in Variables) + { + if (!Decorations.TryGetValue(id, out Dictionary? d)) continue; + if (d.TryGetValue(DecorationDescriptorSet, out uint s) && s == set && + d.TryGetValue(DecorationBinding, out uint b) && b == binding) + { + return Pointers[pointer].Type; + } + } + return null; + } + + /// The block types of every variable in . + public List BlocksIn(uint storage) + { + var blocks = new List(); + foreach ((uint _, (uint pointer, uint variableStorage)) in Variables) + { + if (variableStorage == storage) blocks.Add(Pointers[pointer].Type); + } + return blocks; + } + + public int MemberCount(uint structType) => Structs[structType].Length; + + public uint MemberOffset(uint structType, int member) => + MemberDecorations[(structType, (uint)member)][DecorationOffset]; + + public string? MemberName(uint structType, int member) => + MemberNames.TryGetValue((structType, (uint)member), out string? name) ? name : null; + + public uint? ArrayStride(uint structType, int member) + { + uint type = Structs[structType][member]; + if (!ArrayElements.ContainsKey(type)) return null; + return Decorations.TryGetValue(type, out Dictionary? d) && + d.TryGetValue(DecorationArrayStride, out uint stride) ? stride : null; + } + + /// Specialization constant id to its scalar type name. + public Dictionary SpecIds() + { + var ids = new Dictionary(); + foreach ((uint id, Dictionary d) in Decorations) + { + if (d.TryGetValue(DecorationSpecId, out uint specId) && SpecConstantTypes.TryGetValue(id, out uint type)) + { + ids[specId] = ScalarTypes[type]; + } + } + return ids; + } + + private static Dictionary Get(Dictionary> map, uint id) + { + if (!map.TryGetValue(id, out Dictionary? value)) map[id] = value = new Dictionary(); + return value; + } + + private static string ReadString(ReadOnlySpan words) + { + var bytes = new List(); + foreach (uint word in words) + { + for (int shift = 0; shift < 32; shift += 8) + { + byte b = (byte)(word >> shift); + if (b == 0) return Encoding.UTF8.GetString(bytes.ToArray()); + bytes.Add(b); + } + } + return Encoding.UTF8.GetString(bytes.ToArray()); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/ShaderDeliveryTests.cs b/Optimum.Render.Vulkan.Tests/ShaderDeliveryTests.cs new file mode 100644 index 00000000..cafccec9 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ShaderDeliveryTests.cs @@ -0,0 +1,390 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Security.Cryptography; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Xunit; + +namespace Optimum.Render.Vulkan.Tests; + +public sealed class ShaderDeliveryTests : IDisposable +{ + private readonly string root = Path.Combine(Path.GetTempPath(), "optimum-shader-delivery-" + Guid.NewGuid().ToString("N")); + private string Source => Path.Combine(root, "src"); + private string Output => Path.Combine(root, "out"); + private string Package => Path.Combine(Output, NativeShaderManifest.DirectoryName); + + public ShaderDeliveryTests() + { + Directory.CreateDirectory(Path.Combine(Source, "include")); + File.Copy(Path.Combine(ShaderCorpus.RepositoryRoot, SetConvention.IncludePath), Path.Combine(Source, "include", "bindings.glsl")); + } + + public void Dispose() { if (Directory.Exists(root)) Directory.Delete(root, true); } + + private const string Program = """ + #version 450 + #extension GL_EXT_scalar_block_layout : require + #include "bindings.glsl" + layout(push_constant, scalar) uniform Draw { vec4 tint; } draw; + layout(set = OPTIMUM_SET_STORAGE, binding = OPTIMUM_BINDING_PROGRAM_RECORD, scalar) + uniform Record { float gain; } record; + #if defined(OPTIMUM_VERTEX) + layout(location = 0) out vec2 uv; + void main() { uv = draw.tint.xy; gl_Position = vec4(uv, 0, 1); } + #elif defined(OPTIMUM_FRAGMENT) + #include "color.glsl" + layout(location = 0) in vec2 uv; + layout(location = 0) out vec4 color; + #if TAAMOTION == 1 + layout(location = 1) out vec4 motion; + #endif + void main() { + color = fixtureColor() * draw.tint * record.gain + vec4(uv, 0, 0); + #if TAAMOTION == 1 + motion = vec4(uv, 0, 1); + #endif + } + #endif + """; + + private void Fixtures() + { + File.WriteAllText(Path.Combine(Source, "fixture.glsl"), Program); + File.WriteAllText(Path.Combine(Source, "include", "color.glsl"), "vec4 fixtureColor() { return vec4(0.2, 0.3, 0.4, 1); }"); + } + + private static NativeShaderBuildResult Build(ShaderCompiler compiler, string source) + { + var result = new NativeShaderBuilder(compiler).Build(source); + Assert.True(result.Success, string.Join(Environment.NewLine, result.Errors)); + return result; + } + + [Fact] + public void SingleSourceStagesProduceVariantsWithTheDeclaredAbi() + { + Fixtures(); + using var compiler = new ShaderCompiler(); + var build = Build(compiler, Source); + var program = Assert.Single(build.Manifest.Programs); + Assert.Equal(new[] { "TAAMOTION=0", "TAAMOTION=1" }, program.Variants.Select(v => v.Key)); + foreach (var variant in program.Variants) + { + Assert.Equal(16, variant.Push!.Size); + Assert.Equal(4, variant.Record!.Size); + Assert.Equal(new[] { 0 }, variant.Push.Members.Select(m => m.Offset)); + Assert.Equal(variant.Key.EndsWith("1") ? 2 : 1, variant.FragmentOutputs.Count); + Assert.Equal(new[] { "vertex", "fragment" }.Order(), variant.Stages.Select(s => s.Stage).Order()); + foreach (var stage in variant.Stages) + { + Assert.Equal("fixture.glsl", stage.Source); + Assert.Equal(stage.Sha256, Convert.ToHexStringLower(SHA256.HashData(build.Files[stage.Spirv]))); + Assert.Equal(0x07230203u, BitConverter.ToUInt32(build.Files[stage.Spirv])); + } + } + } + + [Fact] + public void IncludeEditsInvalidateOutputsAndRemovedProgramsLeaveNoStaleBinaries() + { + Fixtures(); + using var compiler = new ShaderCompiler(); + var initial = Build(compiler, Source); + NativeShaderBuilder.Write(initial, Output); + var times = Directory.GetFiles(Package).ToDictionary(Path.GetFileName, File.GetLastWriteTimeUtc); + var unchanged = Build(compiler, Source); + NativeShaderBuilder.Write(unchanged, Output); + Assert.Empty(NativeShaderBuilder.Compare(unchanged, Output)); + Assert.All(Directory.GetFiles(Package), path => Assert.Equal(times[Path.GetFileName(path)], File.GetLastWriteTimeUtc(path))); + + File.WriteAllText(Path.Combine(Source, "include", "color.glsl"), "vec4 fixtureColor() { return vec4(0.8, 0.7, 0.6, 1); }"); + var changed = Build(compiler, Source); + Assert.NotEmpty(NativeShaderBuilder.Compare(changed, Output)); + foreach (var variant in changed.Manifest.Programs[0].Variants) + { + var fragment = variant.Stages.Single(s => s.Stage == "fragment"); + Assert.False(initial.Files[fragment.Spirv].SequenceEqual(changed.Files[fragment.Spirv])); + } + NativeShaderBuilder.Write(changed, Output); + File.Delete(Path.Combine(Source, "fixture.glsl")); + NativeShaderBuilder.Write(Build(compiler, Source), Output); + Assert.Empty(Directory.GetFiles(Package, "*.spv")); + Assert.Empty(NativeShaderManifest.Load(Path.Combine(Package, NativeShaderManifest.FileName)).Programs); + } + + [Fact] + public void ShippedProgramsCompilePackageAndLoadEveryDeclaredStage() + { + using var compiler = new ShaderCompiler(); + string source = Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"); + var build = Build(compiler, source); + string[] names = Directory.GetFiles(source, "*.glsl").Select(Path.GetFileNameWithoutExtension).Order().ToArray()!; + Assert.NotEmpty(names); + Assert.Equal(names, build.Manifest.Programs.Select(p => p.Name).Order()); + NativeShaderBuilder.Write(build, Output); + var library = NativeShaderLibrary.Load(Package, compiler.Identity, out string reason); + Assert.True(library != null, reason); + var sharedBindings = SharedBindings(); + foreach (var program in build.Manifest.Programs) + foreach (var variant in program.Variants) + foreach (var stage in variant.Stages) + { + Assert.True(library!.TryGetSpirv(stage, out byte[] bytes, out string error), error); + Assert.Equal(build.Files[stage.Spirv], bytes); + var reflected = SpirvReflection.Reflect(bytes); + foreach (var binding in reflected.Bindings) + Assert.Contains((binding.Set, binding.Binding, binding.Kind), sharedBindings); + } + Assert.Empty(NativeShaderBuilder.Compare(build, Output)); + } + + [Fact] + public void CorruptPackagedModulesAreRejectedBeforeLinking() + { + Fixtures(); + using var compiler = new ShaderCompiler(); + var build = Build(compiler, Source); + NativeShaderBuilder.Write(build, Output); + var stage = build.Manifest.Programs[0].Variants[0].Stages[0]; + File.WriteAllBytes(Path.Combine(Package, stage.Spirv), new byte[] { 1, 2, 3 }); + var library = NativeShaderLibrary.Load(Package, compiler.Identity, out _); + Assert.NotNull(library); + Assert.False(library!.TryGetSpirv(stage, out _, out string reason)); + Assert.NotEmpty(reason); + Assert.NotEmpty(NativeShaderBuilder.Compare(build, Output)); + } + + [Fact] + public void InvalidSourcesFailWithoutReplacingTheLastGoodPackage() + { + Fixtures(); + using var compiler = new ShaderCompiler(); + var good = Build(compiler, Source); + NativeShaderBuilder.Write(good, Output); + File.WriteAllText(Path.Combine(Source, "fixture.glsl"), Program.Replace("void main()", "void broken main()")); + using var output = new StringWriter(); + using var error = new StringWriter(); + Assert.NotEqual(0, NativeShaderTool.Run(new[] { "--build", Source, Output }, output, error)); + Assert.NotEmpty(error.ToString()); + Assert.Empty(NativeShaderBuilder.Compare(good, Output)); + } + + [Fact] + public void BinaryCacheReusesCompiledModulesAndRejectsCorruption() + { + var cache = new ShaderBinaryCache(Path.Combine(root, "cache")); + using var compiler = new ShaderCompiler { BinaryCache = cache }; + const string source = "#version 450\nvoid main() { gl_Position = vec4(0, 0, 0, 1); }"; + var first = compiler.Compile(source, "fixture.vert", EnumShaderType.VertexShader); + Assert.True(first.Success, first.Error); + var second = compiler.Compile(source, "fixture.vert", EnumShaderType.VertexShader); + Assert.True(second.Success, second.Error); + Assert.Equal(1, cache.Hits); + Assert.Equal(first.Spirv, second.Spirv); + byte[] stored = ShaderBinaryCache.Wrap(first.Spirv); + stored[^1] ^= 1; + Assert.Null(ShaderBinaryCache.Unwrap(stored)); + string key = ShaderBinaryCache.KeyFor(source, EnumShaderType.VertexShader, compiler.Identity); + Assert.NotEqual(key, ShaderBinaryCache.KeyFor(source, EnumShaderType.FragmentShader, compiler.Identity)); + Assert.NotEqual(key, ShaderBinaryCache.KeyFor(source, EnumShaderType.VertexShader, "different-compiler")); + } + + [Fact] + public void DriverCacheRequiresMatchingHardwareDriverAndIntactPayload() + { + var id = new PipelineCacheIdentity(0x8086, 0x4688, 1017088, Enumerable.Range(0, 16).Select(i => (byte)i).ToArray()); + var blob = new byte[64]; + BitConverter.TryWriteBytes(blob.AsSpan(0), 32u); + BitConverter.TryWriteBytes(blob.AsSpan(4), 1u); + BitConverter.TryWriteBytes(blob.AsSpan(8), id.VendorId); + BitConverter.TryWriteBytes(blob.AsSpan(12), id.DeviceId); + id.Uuid.CopyTo(blob, 16); + string path = PipelineCacheFile.PathFor(root, id); + Assert.True(PipelineCacheFile.Save(path, blob, id)); + Assert.Equal(blob, PipelineCacheFile.Load(path, id)); + foreach (var other in new[] + { + new PipelineCacheIdentity(0x10de, id.DeviceId, id.DriverVersion, id.Uuid), + new PipelineCacheIdentity(id.VendorId, id.DeviceId + 1, id.DriverVersion, id.Uuid), + new PipelineCacheIdentity(id.VendorId, id.DeviceId, id.DriverVersion + 1, id.Uuid), + new PipelineCacheIdentity(id.VendorId, id.DeviceId, id.DriverVersion, new byte[16]), + }) Assert.Null(PipelineCacheFile.Load(path, other)); + byte[] stored = File.ReadAllBytes(path); + File.WriteAllBytes(path, stored[..^1]); + Assert.Null(PipelineCacheFile.Load(path, id)); + stored[^1] ^= 1; + File.WriteAllBytes(path, stored); + Assert.Null(PipelineCacheFile.Load(path, id)); + } + + [Fact] + public void FailedAtomicCacheReplacementPreservesTheLastGoodFile() + { + string path = Path.Combine(root, "blocked-cache.bin"); + File.WriteAllBytes(path, new byte[] { 9, 8, 7 }); + Assert.False(CacheFileWriter.WriteAtomically(path, new byte[] { 1 }, + (_, _) => throw new IOException("sharing violation"), _ => { })); + Assert.Equal(new byte[] { 9, 8, 7 }, File.ReadAllBytes(path)); + Assert.Empty(Directory.GetFiles(root, "*.tmp")); + Assert.True(CacheFileWriter.WriteAtomically(path, new byte[] { 2 })); + Assert.Equal(new byte[] { 2 }, File.ReadAllBytes(path)); + } + + [Fact] + public void PipelineWarmupRecordsStaySeparatedBySettingsAndProgram() + { + var log = new PipelineKeyLog(); + PipelineKeyLogEntry Entry(ulong settings, ulong shader) => new() + { + SettingsHash = settings, ProgramHash = new UInt128(0, shader), + Bindings = Array.Empty(), Attributes = Array.Empty(), + DepthFormat = Silk.NET.Vulkan.Format.Undefined, + PolygonMode = Silk.NET.Vulkan.PolygonMode.Fill, Topology = Silk.NET.Vulkan.PrimitiveTopology.TriangleList, + ColorFormats = new[] { Silk.NET.Vulkan.Format.R8G8B8A8Unorm }, + Blend = new[] { AttachmentBlend.Default }, + }; + log.Record(Entry(1, 20), 100); + log.Record(Entry(2, 20), 200); + log.Record(Entry(1, 21), 300); + log.Record(Entry(1, 20), 400); + string path = Path.Combine(root, "warmup.keys"); + Assert.True(log.Save(path)); + var loaded = PipelineKeyLog.Load(path); + Assert.Equal(3, loaded.Count); + Assert.Equal(400, Assert.Single(loaded.Matching(1, new UInt128(0, 20))).LastSeenUnixMs); + Assert.Equal(200, Assert.Single(loaded.Matching(2, new UInt128(0, 20))).LastSeenUnixMs); + Assert.Empty(loaded.Matching(2, new UInt128(0, 21))); + File.WriteAllText(path, "invalid partial write"); + Assert.Equal(0, PipelineKeyLog.Load(path).Count); + } + + private static HashSet<(int Set, int Binding, SpirvDescriptorKind Kind)> SharedBindings() + { + static SpirvDescriptorKind Kind(Silk.NET.Vulkan.DescriptorType type) => type switch { + Silk.NET.Vulkan.DescriptorType.UniformBuffer or Silk.NET.Vulkan.DescriptorType.UniformBufferDynamic => SpirvDescriptorKind.UniformBuffer, + Silk.NET.Vulkan.DescriptorType.StorageBuffer or Silk.NET.Vulkan.DescriptorType.StorageBufferDynamic => SpirvDescriptorKind.StorageBuffer, + _ => SpirvDescriptorKind.CombinedImageSampler, + }; + return SharedPipelineLayout.FrameBindings().Select(b => (SetConvention.FrameSet, (int)b.Binding, Kind(b.DescriptorType))) + .Concat(SharedPipelineLayout.StorageBindings().Select(b => (SetConvention.StorageSet, (int)b.Binding, Kind(b.DescriptorType)))) + .Concat(SetConvention.TextureArrays.Select(b => (SetConvention.TextureSet, b.Value, SpirvDescriptorKind.CombinedImageSampler))).ToHashSet(); + } + + [Fact] + public void ReflectionPreservesDeclaredAbiAndFindsActualUsesInOptimizedCode() + { + const string source = """ + #version 450 + #extension GL_EXT_scalar_block_layout : require + #extension GL_EXT_nonuniform_qualifier : require + layout(set=0, binding=1) uniform sampler2DShadow unusedShadow; + layout(set=1, binding=2) uniform sampler2D images[]; + layout(set=2, binding=4, std430) readonly buffer Data { uint words[]; } data; + layout(push_constant, scalar) uniform Draw { uint index; vec3 origin; mat4 transform; } draw; + layout(set=2, binding=3, scalar) uniform Record { vec2 scale; float weights[3]; mat3 normal; int kind; } record; + layout(constant_id=3) const int MODE = 2; + layout(constant_id=7) const float THRESHOLD = 0.25; + layout(constant_id=9) const bool ENABLE = true; + layout(constant_id=11) const uint COUNT = 5u; + layout(location=0) in vec2 uv; + layout(location=1) flat in ivec3 flags; + layout(location=0) out vec4 color; + layout(location=2) out vec4 unwritten; + layout(location=3) out vec4 partial; + void main() { + color = texture(images[draw.index], uv * record.scale) + vec4(record.normal * draw.origin, 1) * record.weights[2]; + color *= float(data.words[flags.x]) + float(MODE) + THRESHOLD + float(COUNT); + if (ENABLE) partial.xy = draw.origin.xy; + } + """; + using var compiler = new ShaderCompiler(); + var declaredCode = compiler.CompileForReflection(source, "abi.frag", EnumShaderType.FragmentShader); + var shippedCode = compiler.Compile(source, "abi.frag", EnumShaderType.FragmentShader); + Assert.True(declaredCode.Success, declaredCode.Error); Assert.True(shippedCode.Success, shippedCode.Error); + var declared = SpirvReflection.Reflect(declaredCode.Spirv); + var shipped = SpirvReflection.Reflect(shippedCode.Spirv); + Assert.Equal("main", declared.EntryPoint); + Assert.Equal(SpirvReflection.ExecutionModelFragment, declared.ExecutionModel); + Assert.Equal(new[] { (0, "vec2"), (1, "ivec3") }, declared.Inputs.Select(v => (v.Location, v.GlslType))); + Assert.Equal(new[] { 0, 2, 3 }, declared.Outputs.Select(v => v.Location)); + Assert.Equal(new[] { ("index", 0, 4), ("origin", 4, 12), ("transform", 16, 64) }, + declared.PushConstants!.Members.Select(m => (m.Name, m.Offset, m.Size))); + Assert.Equal(80, declared.PushConstants.Size); + var record = declared.Bindings.Single(b => b.Set == 2 && b.Binding == 3); + Assert.Equal(SpirvDescriptorKind.UniformBuffer, record.Kind); + Assert.Equal(new[] { ("scale", 0, 8, 0), ("weights", 8, 12, 3), ("normal", 20, 36, 0), ("kind", 56, 4, 0) }, + record.Block!.Members.Select(m => (m.Name, m.Offset, m.Size, m.ArrayLength))); + Assert.Equal(60, record.Block.Size); + Assert.True(declared.Bindings.Single(b => b.Set == 1).RuntimeArray); + var storage = declared.Bindings.Single(b => b.Binding == 4); + Assert.Equal(SpirvDescriptorKind.StorageBuffer, storage.Kind); + Assert.Equal(-1, Assert.Single(storage.Block!.Members).ArrayLength); + Assert.Equal(new[] { (3, "int", 2.0), (7, "float", 0.25), (9, "bool", 1.0), (11, "uint", 5.0) }, + declared.SpecConstants.Select(c => (c.SpecId, c.GlslType, c.DefaultValue))); + Assert.Equal(new[] { 0, 3 }, shipped.WrittenOutputLocations); + Assert.Equal(new[] { (1, 2), (2, 3), (2, 4) }, shipped.Bindings.Where(b => shipped.UsedVariables.Contains(b.VariableId)).Select(b => (b.Set, b.Binding))); + Assert.Equal(new[] { (1, 2) }, shipped.PushMemberIndexes[0]); + Assert.All(shipped.Bindings, binding => Assert.Empty(binding.Name)); + Assert.Throws(() => SpirvReflection.Reflect(new byte[] { 1, 2, 3 })); + Assert.Throws(() => SpirvReflection.Reflect(new byte[20])); + + var vertex = compiler.CompileForReflection(""" + #version 450 + layout(location=0) in vec3 position; + layout(location=1) in mat2 packedIn; + layout(location=3) in ivec4 color; + layout(location=0) out vec2 uv; + void main() { uv = packedIn[0] + vec2(color.xy); gl_Position = vec4(position, 1); } + """, "abi.vert", EnumShaderType.VertexShader); + Assert.True(vertex.Success, vertex.Error); + var input = SpirvReflection.Reflect(vertex.Spirv); + Assert.Equal(SpirvReflection.ExecutionModelVertex, input.ExecutionModel); + Assert.Equal(new[] { (0, "vec3"), (1, "mat2"), (3, "ivec4") }, input.Inputs.Select(v => (v.Location, v.GlslType))); + Assert.Equal("uv", Assert.Single(input.Outputs).Name); + } + + [Fact] + public void CompiledSharedIncludesMatchCpuFrameOffsetsAndSpecializationIds() + { + using var compiler = new ShaderCompiler(); + string sum = string.Join(" + ", SpecializationConvention.Constants.Select(c => "float(" + c.Name + ")")); + string probe = "#version 450\n#include \"frame.glsl\"\n#include \"specialization.glsl\"\n" + + "layout(location=0) out vec4 color;\nvoid main() { color = vec4(optimumFrame.zNear + optimumFrame.pointLights[99].x + " + sum + "); }"; + var compiled = NativeShaderTree.Compile(compiler, probe, EnumShaderType.FragmentShader, "shared-abi"); + Assert.True(compiled.Success, compiled.Error); + var reflected = SpirvReflection.Reflect(compiled.Spirv); + var block = reflected.Bindings.Single(b => b.Set == FrameGlobals.Set && b.Binding == FrameGlobals.Binding).Block!; + Assert.Equal(FrameGlobals.BlockSize, block.Size); + Assert.Equal(FrameGlobals.Members.Select(m => (m.Offset, m.Size, m.ArrayLength)), + block.Members.Select(m => (m.Offset, m.Size, m.ArrayLength))); + Assert.Equal(SpecializationConvention.Constants.Select(c => ((int)c.Id, c.GlslType, double.Parse(c.Default, System.Globalization.CultureInfo.InvariantCulture))).Order(), + reflected.SpecConstants.Select(c => (c.SpecId, c.GlslType, c.DefaultValue)).Order()); + foreach (var binding in reflected.Bindings) + Assert.Contains((binding.Set, binding.Binding, binding.Kind), SharedBindings()); + } + + [Fact] + public void OnlyCompatibleUniformsFromTheirOwningIncludesEnterTheSharedBlock() + { + const string code = "uniform vec3 pointLights[4]; uniform float viewDistance; uniform vec3 lightPosition; uniform float tint; void main() {}"; + ProgramInterfaceLayout Layout(string source, IReadOnlySet? owners) => ProgramInterfaceLayout.Build( + new[] { (EnumShaderType.VertexShader, GlslParser.Parse(source)) }, null, owners); + var owned = Layout(code, new HashSet { "fogandlight.vsh" }); + Assert.Equal(new[] { "pointLights", "viewDistance" }, owned.FrameMemberDeclaredLengths.Keys.Order()); + Assert.Equal(new[] { "lightPosition", "tint" }, owned.Members.Select(m => m.Name).Order()); + Assert.Equal(4, owned.FrameMemberDeclaredLengths["pointLights"]); + Assert.False(Layout(code, null).UsesFrameBlock); + Assert.False(Layout("uniform float pointLights[4]; void main() {}", new HashSet { "fogandlight.vsh" }).UsesFrameBlock); + Assert.False(Layout("uniform vec3 pointLights[101]; void main() {}", new HashSet { "fogandlight.vsh" }).UsesFrameBlock); + byte[] defaults = FrameGlobals.CreateShadow(); + foreach (var pair in new[] { ("zNear", 0.3f), ("zFar", 1500f), ("windWaveIntensity", 1f) }) + { + Assert.True(FrameGlobals.TryGetMember(pair.Item1, out var member)); + Assert.Equal(pair.Item2, BitConverter.ToSingle(defaults, member.Offset)); + } + } +} diff --git a/Optimum.Render.Vulkan.Tests/ShaderTranslationTests.cs b/Optimum.Render.Vulkan.Tests/ShaderTranslationTests.cs new file mode 100644 index 00000000..e1fd6c1a --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/ShaderTranslationTests.cs @@ -0,0 +1,500 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// The gate on the whole backend: every shader the client actually loads has to +/// survive translation to SPIR-V. +/// +/// If a shader cannot be translated automatically it would have to be +/// hand-ported, and hand-porting does not scale to mod shaders, which are GLSL +/// authored by third parties and only exist at runtime. So this is not a +/// nice-to-have test - a failure here means the approach does not hold. +/// +public class ShaderTranslationTests +{ + private readonly ITestOutputHelper _output; + + public ShaderTranslationTests(ITestOutputHelper output) => _output = output; + + [SkippableFact] + public void EveryVanillaProgramTranslatesToSpirv() + { + Skip.If(ShaderCorpus.AssetRoot == null, + "No bootstrapped game assets; run scripts/bootstrap.sh or set VINTAGE_STORY_ASSETS."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var programs = ShaderCorpus.ProgramNames(files); + + Assert.NotEmpty(programs); + + using var compiler = new ShaderCompiler(); + var failures = new List(); + int translated = 0; + + foreach (ShaderCorpus.ShaderVariant variant in ShaderCorpus.Variants()) + { + foreach (string program in programs) + { + var stages = ShaderCorpus.BuildProgram(program, files, includes, variant); + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + + if (result.Success) + { + translated++; + continue; + } + + failures.Add($"[{variant.Name}] {program}: {string.Join("; ", result.Errors)}"); + } + } + + _output.WriteLine($"{translated} program/variant combinations translated, {failures.Count} failed."); + + if (failures.Count > 0) + { + var report = new StringBuilder(); + report.Append(failures.Count).Append(" shader program(s) failed to translate:\n"); + foreach (string failure in failures.Take(40)) + { + report.Append(" ").Append(failure).Append('\n'); + } + Assert.Fail(report.ToString()); + } + } + + /// + /// The corpus rows above all carry USEOIT 1 and no ALLOWDEPTHOFFSET, because + /// those are the settings every program shares. The TAA motion writers live + /// in exactly the configurations they leave out: + /// + /// - entityanimated's writer is inside `#if USEOIT == 0`, which only the + /// opaque Entityanimated registration and ModSystemFpHands' hand shader + /// produce, so the corpus has never translated the entity writer at all - + /// including its second AnimationPrev uniform block, the only place the + /// backend meets two named blocks in one program; + /// - the two-argument writer that stamps `gl_FragCoord.z + depthOffset` into + /// the motion alpha only exists when ALLOWDEPTHOFFSET is stamped, which + /// ModSystemFpHands does for its private copies of entityanimated and + /// standard - the first-person hands and the first-person item. + /// + /// Those are shipped configurations, so they belong in the translation gate. + /// + [SkippableFact] + public void MotionWritersTranslateInTheConfigurationsTheClientReallyBuilds() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + + // (program, variant) pairs the client produces with TAA on. + var cases = new List<(string Program, ShaderCorpus.ShaderVariant Variant)>(); + foreach (int ssao in new[] { 0, 2 }) + { + int location = ssao > 0 ? 4 : 2; + + cases.Add(("entityanimated", new ShaderCorpus.ShaderVariant + { + Name = $"entity-opaque-ssao{ssao}", + UseOit = 0, SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, + TaaMotion = 1, TaaMotionLocation = location, + })); + cases.Add(("entityanimated", new ShaderCorpus.ShaderVariant + { + Name = $"entity-fphands-ssao{ssao}", + UseOit = 0, SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, + TaaMotion = 1, TaaMotionLocation = location, + ExtraPrefix = "#define ALLOWDEPTHOFFSET 1", + })); + cases.Add(("standard", new ShaderCorpus.ShaderVariant + { + Name = $"standard-fpitem-ssao{ssao}", + SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, + TaaMotion = 1, TaaMotionLocation = location, + ExtraPrefix = "#define ALLOWDEPTHOFFSET 1", + })); + // TAA P4 review: the decal writer's SSBO branch. USESSBO tracks + // ScreenManager.Platform.UseSSBOs, which is on by default, and it is + // the branch where vertexPos and renderFlagsIn are locals unpacked + // from the face buffer rather than vertex attributes - so the + // previous-position block reads different symbols there. The corpus + // rows that carry USESSBO 1 all carry TAAMOTION 0, so the + // combination the client really ships was outside the gate. + cases.Add(("decals", new ShaderCorpus.ShaderVariant + { + Name = $"decals-ssbo-ssao{ssao}", + SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, UseSsbo = 1, + TaaMotion = 1, TaaMotionLocation = location, + })); + cases.Add(("decals", new ShaderCorpus.ShaderVariant + { + Name = $"decals-nossbo-ssao{ssao}", + SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, UseSsbo = 0, + TaaMotion = 1, TaaMotionLocation = location, + })); + // TAA P4 review: the cube-particle writer's VEC3SCALE branch. + // VSEssentials' EntityParticleSystem stamps `#define VEC3SCALE 1` on + // its private copy of particlescube (EntityParticleSystem.cs:190), + // which is the per-axis-scale position path - a second place the + // twin previous-position function has to agree with vanilla's own + // lines. No corpus row produces it. + cases.Add(("particlescube", new ShaderCorpus.ShaderVariant + { + Name = $"particlescube-vec3scale-ssao{ssao}", + SsaoLevel = ssao, DynLights = 4, ShadowQuality = 2, + TaaMotion = 1, TaaMotionLocation = location, + ExtraPrefix = "#define VEC3SCALE 1", + })); + } + + using var compiler = new ShaderCompiler(); + var failures = new List(); + + foreach ((string program, ShaderCorpus.ShaderVariant variant) in cases) + { + var stages = ShaderCorpus.BuildProgram(program, files, includes, variant); + Assert.NotEmpty(stages); + + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + if (!result.Success) + { + failures.Add($"[{variant.Name}] {program}: {string.Join("; ", result.Errors)}"); + continue; + } + + foreach (KeyValuePair stage in result.Spirv) + { + Assert.True(stage.Value.Length >= 20 && stage.Value.Length % 4 == 0, + $"{variant.Name} {program} {stage.Key}: malformed SPIR-V"); + Assert.Equal(0x07230203u, BitConverter.ToUInt32(stage.Value, 0)); + } + } + + Assert.True(failures.Count == 0, string.Join("\n", failures)); + } + + /// + /// The writer only means anything if it is actually in the translated source. + /// A define typo or a stray guard would leave every assertion above passing + /// on a shader that emits no motion at all. + /// + [SkippableFact] + public void TheEntityMotionWriterSurvivesThePreprocessorInTheOpaqueConfiguration() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var variant = new ShaderCorpus.ShaderVariant + { + Name = "entity-opaque", UseOit = 0, SsaoLevel = 2, DynLights = 4, + TaaMotion = 1, TaaMotionLocation = 4, + ExtraPrefix = "#define ALLOWDEPTHOFFSET 1", + }; + + var stages = ShaderCorpus.BuildProgram("entityanimated", files, includes, variant); + + using var compiler = new ShaderCompiler(); + + // The raw Code still carries every #if branch, so asserting on it would + // pass even when TAAMOTION or USEOIT compile the writer out. Only the + // preprocessed text says what the compiler actually sees. + string vertex = Preprocess(compiler, stages, EnumShaderType.VertexShader); + Assert.Contains("PrevElementTransforms", vertex); + Assert.Contains("previousWarpState()", vertex); + Assert.Contains("applyVertexWarpingState", vertex); + + string fragment = Preprocess(compiler, stages, EnumShaderType.FragmentShader); + Assert.Contains("outMotion", fragment); + Assert.Contains("gl_FragCoord.z + depthOffset", fragment); + } + + /// Runs one stage through the real preprocessor and returns its text. + private static string Preprocess( + ShaderCompiler compiler, IReadOnlyList stages, EnumShaderType stage) + { + ShaderStageSource source = stages.Single(s => s.Stage == stage); + ShaderCompileResult result = + compiler.Preprocess(source.Code, source.PrefixCode, source.Filename, source.Stage); + + Assert.True(result.Success, $"{stage}: {result.Error}"); + return result.PreprocessedText; + } + + [SkippableFact] + public void TranslatedProgramsProduceValidSpirvForEveryStage() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var variant = ShaderCorpus.Variants().First(v => v.Name == "everything-on"); + + using var compiler = new ShaderCompiler(); + + foreach (string program in ShaderCorpus.ProgramNames(files)) + { + var stages = ShaderCorpus.BuildProgram(program, files, includes, variant); + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + + Assert.True(result.Success, $"{program}: {string.Join("; ", result.Errors)}"); + Assert.Equal(stages.Count, result.Spirv.Count); + + foreach (KeyValuePair stage in result.Spirv) + { + byte[] spirv = stage.Value; + Assert.True(spirv.Length >= 20, $"{program} {stage.Key}: SPIR-V too short"); + Assert.True(spirv.Length % 4 == 0, $"{program} {stage.Key}: SPIR-V not word-aligned"); + + // 0x07230203 is the SPIR-V magic number. + uint magic = BitConverter.ToUInt32(spirv, 0); + Assert.True(magic == 0x07230203u, + $"{program} {stage.Key}: bad SPIR-V magic 0x{magic:X8}"); + } + } + } + + /// + /// The chunk shaders are the ones that would hurt most to hand-port: they + /// carry the SSBO vertex-fetch path, Optimum's greedy-mesh decode, and the + /// heaviest include graph in the game. + /// + [SkippableFact] + public void ChunkShadersTranslateWithSsboAndGreedyMeshEnabled() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var variant = ShaderCorpus.Variants().First(v => v.Name == "everything-on"); + + using var compiler = new ShaderCompiler(); + + foreach (string program in new[] { "chunkopaque", "chunktransparent", "chunktopsoil", "chunkliquid" }) + { + Skip.IfNot(files.ContainsKey(program + ".vsh"), $"{program} not present"); + + var stages = ShaderCorpus.BuildProgram(program, files, includes, variant); + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + + Assert.True(result.Success, $"{program}: {string.Join("; ", result.Errors)}"); + + // The storage buffer moves to FaceData's binding in set 2, whatever the + // shader declared: the mesh path binds the vertex buffer there. + if (program is "chunkopaque" or "chunktransparent" or "chunktopsoil") + { + BlockBinding? faceData = result.Layout.StorageBlocks + .FirstOrDefault(b => b.BlockName == "faceDataBuf"); + Assert.NotNull(faceData); + Assert.Equal(SetConvention.FaceDataBinding, faceData!.Binding); + } + } + } + + /// + /// final.fsh declares "uniform float extraGamma = 1.0;" and never assigns it + /// unless colour grading is active. GL applies declared defaults at link time, + /// so the shadow buffer has to start out carrying them or the screen comes + /// back black. + /// + [SkippableFact] + public void DeclaredUniformDefaultsReachTheShadowBuffer() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var variant = ShaderCorpus.Variants().First(v => v.Name == "everything-on"); + + using var compiler = new ShaderCompiler(); + var stages = ShaderCorpus.BuildProgram("final", files, includes, variant); + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + + Assert.True(result.Success, string.Join("; ", result.Errors)); + + UniformMember member = result.Layout.MembersByName["extraGamma"]; + Assert.Equal("1.0", member.Initializer); + + byte[] shadow = result.Layout.CreateShadowBuffer(); + Assert.Equal(1.0f, BitConverter.ToSingle(shadow, member.Offset), 5); + } + + /// + /// A uniform named in both stages is one uniform in GL. zNear and zFar come + /// from the fogandlight includes and appear on both sides. + /// + [SkippableFact] + public void UniformsSharedBetweenStagesGetOneSlot() + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + + var files = ShaderCorpus.LoadShaderFiles(); + var includes = ShaderCorpus.LoadIncludes(); + var variant = ShaderCorpus.Variants().First(v => v.Name == "everything-on"); + + using var compiler = new ShaderCompiler(); + var stages = ShaderCorpus.BuildProgram("chunkopaque", files, includes, variant); + TranslatedProgram result = ShaderTranslator.Translate(stages, compiler); + + Assert.True(result.Success, string.Join("; ", result.Errors)); + + var names = result.Layout.Members.Select(m => m.Name).ToList(); + Assert.Equal(names.Count, names.Distinct(StringComparer.Ordinal).Count()); + + // Offsets must not overlap. + var ordered = result.Layout.Members.OrderBy(m => m.Offset).ToList(); + for (int i = 1; i < ordered.Count; i++) + { + Assert.True(ordered[i].Offset >= ordered[i - 1].Offset + ordered[i - 1].Size, + $"'{ordered[i].Name}' overlaps '{ordered[i - 1].Name}'"); + } + } + + // Small synthetic programs cover translator edges absent from the shipped shader corpus. + private static ProgramInterfaceLayout Layout(params (EnumShaderType Stage, string Code)[] stages) => + ProgramInterfaceLayout.Build(stages.Select(s => (s.Stage, GlslParser.Parse(s.Code))).ToList()); + + [Theory] + [InlineData("layout(location=0) out vec4 value[3]; void main(){ value[1] = vec4(1); }", "1")] + [InlineData("layout(location=0) out vec4 value[3]; void main(){ value[0] = vec4(1); value[2].r = 0; }", "0,2")] + [InlineData("layout(location=0) out vec4 value[3]; uniform int index; void main(){ value[index] = vec4(1); }", "0,1,2")] + [InlineData("layout(location=0) out vec4 value[3]; uniform vec4 src[3]; void main(){ value = src; }", "0,1,2")] + [InlineData("layout(location=0) out vec4 value; void main(){ if (value == vec4(0)) discard; }", "")] + public void FragmentWriteMaskTracksStoresWithoutTreatingReadsAsWrites(string fragment, string written) + { + var layout = Layout((EnumShaderType.FragmentShader, "#version 330 core\n" + fragment)); + Assert.Equal(written, string.Join(",", layout.WrittenFragmentOutputs.OrderBy(i => i))); + } + + [Fact] + public void UniformPackingAndNamedBlocksMatchClientUploadMemory() + { + var layout = Layout((EnumShaderType.VertexShader, """ + #version 330 core + uniform vec3 pointLights[4]; + uniform mat4x3 bones[2]; + uniform float density = 0.75; + layout(std140, binding=0) uniform Lights { vec4 light; }; + layout(std140) uniform AnimationPrev { mat4 previous[2]; }; + layout(std140, binding=3) uniform Animation { mat4 current[2]; }; + void main() {} + """)); + Assert.Empty(layout.Errors); + Assert.Equal((0, 48), (layout.MembersByName["pointLights"].Offset, layout.MembersByName["pointLights"].Size)); + Assert.Equal((48, 96), (layout.MembersByName["bones"].Offset, layout.MembersByName["bones"].Size)); + Assert.Equal(144, layout.MembersByName["density"].Offset); + Assert.Equal(0.75f, BitConverter.ToSingle(layout.CreateShadowBuffer(), 144)); + Assert.Equal(new[] { ("Lights", 4), ("AnimationPrev", SetConvention.AnimationPrevBinding), + ("Animation", SetConvention.AnimationBinding) }, + layout.UniformBlocks.Select(b => (b.BlockName, b.Binding))); + } + + [Fact] + public void BindlessSlotsAndFrameTexturesStayInTheirAssignedSets() + { + const string source = """ + #version 330 core + uniform sampler2DArray terrainTex; + uniform sampler2D glowTex; + uniform sampler2DShadow shadowMapFar; + uniform sampler2D sky; + uniform float alphaTest; + out vec4 color; + void main() { + color = texture(terrainTex, vec3(0.5)) + texture(glowTex, vec2(0.5)) + + texture(shadowMapFar, vec3(0.5)) + texture(sky, vec2(0.5)); + } + """; + var layout = Layout((EnumShaderType.FragmentShader, source)); + Assert.Empty(layout.Errors); + Assert.Equal((TextureKind.Texture2DArray, 0), (layout.SamplersByName["terrainTex"].Kind, layout.SamplersByName["terrainTex"].PushOffset)); + Assert.Equal((TextureKind.Texture2D, 4), (layout.SamplersByName["glowTex"].Kind, layout.SamplersByName["glowTex"].PushOffset)); + Assert.Equal(8, layout.PushConstantSize); + Assert.Equal(1, layout.SamplersByName["shadowMapFar"].FrameBinding); + Assert.Equal(3, layout.SamplersByName["sky"].FrameBinding); + Assert.Equal(-1, layout.SamplersByName["terrainTex"].FrameBinding); + Assert.DoesNotContain("glowTex", layout.MembersByName.Keys); + Assert.Equal(4, layout.BlockSize); + string code = ShaderRewriter.Rewrite(GlslParser.Parse(source), layout, EnumShaderType.FragmentShader, false).Code; + Assert.Contains("texture(optimumTextures2DArray[terrainTex], vec3(0.5))", code); + Assert.Contains("texture(optimumTextures2D[glowTex], vec2(0.5))", code); + Assert.Contains("layout(set = 0, binding = 1) uniform sampler2DShadow shadowMapFar;", code); + Assert.Contains("layout(set = 0, binding = 3) uniform sampler2D sky;", code); + } + + [Theory] + [InlineData("texture(tex, uv)")] + [InlineData("texelFetch(tex, ivec2(0), 0)")] + [InlineData("textureLod(tex, uv, 0.0)")] + [InlineData("textureGather(tex, uv, 1)")] + [InlineData("textureGrad(tex, uv, vec2(0), vec2(0))")] + [InlineData("vec4(textureSize(tex, 0), 0, 1)")] + public void SamplingCallFormsCompileWithIndexedBindlessTextures(string expression) + { + string source = "#version 330 core\nuniform sampler2D tex; in vec2 uv; out vec4 color; void main(){ color = " + expression + "; }"; + var layout = Layout((EnumShaderType.FragmentShader, source)); + string code = ShaderRewriter.Rewrite(GlslParser.Parse(source), layout, EnumShaderType.FragmentShader, false).Code; + Assert.Contains(expression.Replace("(tex,", "(optimumTextures2D[tex],", StringComparison.Ordinal), code); + using var compiler = new ShaderCompiler(); + var result = compiler.Compile(code, "sampling.frag", EnumShaderType.FragmentShader); + Assert.True(result.Success, result.Error); + } + + [Fact] + public void MalformedOrUnsupportedInterfacesAreRejectedBeforePipelineCreation() + { + ProgramInterfaceLayout One(string declaration) => Layout((EnumShaderType.FragmentShader, + "#version 330 core\n" + declaration + "\nvoid main() {}")); + Assert.Contains(One("uniform sampler2D values[4];").Errors, e => e.Contains("array", StringComparison.Ordinal)); + Assert.Contains(One("uniform sampler1D line;").Errors, e => e.Contains("no bindless array", StringComparison.Ordinal)); + Assert.Contains(One(string.Concat(Enumerable.Range(0, 33).Select(i => $"uniform sampler2D s{i};\n"))).Errors, + e => e.Contains("push byte 132", StringComparison.Ordinal)); + Assert.Contains(One(string.Concat(Enumerable.Range(0, 5).Select(i => $"layout(std140) uniform B{i} {{ vec4 v{i}; }};\n"))).Errors, + e => e.Contains("does not fit set 2", StringComparison.Ordinal)); + foreach (string invalid in new[] { "MAX_LIGHTS", "3 *", "", "4 / 0" }) + Assert.False(GlslParser.TryEvaluateConstantInt(invalid, out _)); + Assert.True(GlslParser.TryEvaluateConstantInt("(2 + 3) * 4", out int count)); Assert.Equal(20, count); + } + + [Fact] + public void GeometryEmissionRemapsDepthAtEachCallWithoutChangingControlFlow() + { + const string source = """ + #version 330 core + layout(triangles) in; + layout(triangle_strip, max_vertices = 4) out; + void main() { + for (int i = 0; i < 3; i++) gl_Position = gl_in[i].gl_Position, EmitVertex(); + if (true) EmitVertex (); else EndPrimitive(); + EndPrimitive(); + } + """; + var layout = Layout((EnumShaderType.GeometryShader, source)); + var rewritten = ShaderRewriter.Rewrite(GlslParser.Parse(source), layout, EnumShaderType.GeometryShader, true); + Assert.Empty(rewritten.Errors); + Assert.Contains("gl_Position.z = (gl_Position.z + gl_Position.w) * 0.5; EmitVertex();", rewritten.Code); + Assert.Contains("gl_Position = gl_in[i].gl_Position, _optimum_emit_vertex();", rewritten.Code); + Assert.Contains("if (true) _optimum_emit_vertex (); else EndPrimitive();", rewritten.Code); + } + + [Fact] + public void PreprocessingHandlesOldVersionsAndPrefixWithoutANewline() + { + Assert.StartsWith("#version 450", ShaderCompiler.RaiseVersionForPreprocessing("#version 130\nvoid main() {}")); + Assert.Equal("#version 330 core\nvoid main() {}", ShaderCompiler.RaiseVersionForPreprocessing("#version 330 core\nvoid main() {}")); + Assert.StartsWith("#version 330 core\n#define A 1\n", + ShaderCompiler.SplicePrefix("#version 330 core", "#define A 1\n")); + Assert.True(GlslParser.Parse("#version 330 core\nvoid /* comment */ main() {}").HasMain); + Assert.False(GlslParser.Parse("#version 330 core\nvoid mainImage() {}").HasMain); + } +} diff --git a/Optimum.Render.Vulkan.Tests/SubmissionTests.cs b/Optimum.Render.Vulkan.Tests/SubmissionTests.cs new file mode 100644 index 00000000..382798d1 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/SubmissionTests.cs @@ -0,0 +1,542 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class SubmissionTests(ITestOutputHelper output) +{ + private const string Triangle = """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); } + """; + private const string White = """ + #version 330 core + out vec4 color; + void main() { color = vec4(1); } + """; + private static readonly WaitSite[] NonblockingSites = { + WaitSite.UploadSubmit, WaitSite.FlushFrame, WaitSite.DeviceWaitIdle, + WaitSite.Readback, WaitSite.OcclusionQuery, WaitSite.SwapchainAcquire, WaitSite.Present, + }; + private static long[] WaitCounts() => NonblockingSites.Select(VulkanStats.WaitCount).ToArray(); + + private VulkanContext OpenContext(List messages) => GpuTest.CreateContext(output, messages); + private VulkanDevice OpenDevice() => GpuTest.CreateDevice(output); + + private sealed class Retired(Action destroy) : IDisposable + { + public void Dispose() => destroy(); + } + + [Fact] + public void IndirectRegionsDoNotWrapOrGrowUntilTheirSlotIsRecycled() + { + var ring = new IndirectRing(2, minimumCapacity: 100); + ring.BeginFrame(0, out _); ring.Attach(100); + for (ulong i = 0; i < 5; i++) + { + Assert.True(ring.TryAllocate(20, out ulong offset)); + Assert.Equal(i * 20, offset); + } + for (int i = 0; i < 3; i++) Assert.False(ring.TryAllocate(20, out _)); + Assert.Equal(100UL, ring.CapacityOf(0)); + ring.BeginFrame(1, out _); + Assert.Equal(100UL, ring.CursorOf(0)); + Assert.True(ring.NeedsBuffer(20, out ulong capacity)); + Assert.True(capacity >= 160); + ring.Attach(capacity); Assert.True(ring.TryAllocate(20, out _)); + Assert.True(ring.BeginFrame(0, out capacity)); + ring.Attach(capacity); + for (ulong i = 0; i < 8; i++) + { + Assert.True(ring.TryAllocate(20, out ulong offset)); + Assert.Equal(i * 20, offset); + } + Assert.Equal(20UL, ring.CursorOf(1)); + Assert.Equal(0, ring.OverflowsOf(0)); + } + + [SkippableFact] + public unsafe void FrameSlotsKeepAlignedDisjointStorageUntilTheirLatestSubmissionCompletes() + { + var messages = new List(); + using (var context = OpenContext(messages)) + using (var ring = new FrameRing(context, framesInFlight: 3, uniformRingSize: 65537)) + { + var slots = new FrameSlot[3]; + var allocations = new RingAllocation[3]; + var stamps = new int[3]; + var submitted = new ulong[3]; + int retired = 0; + for (int frame = 0; frame < 18; frame++) + { + int index = frame % 3; + var slot = ring.BeginFrame(); + if (frame >= 3) + { + Assert.Same(slots[index], slot); + Assert.True(ring.Timeline.FrameCompleted >= submitted[index]); + } + else slots[index] = slot; + Assert.Equal(0UL, slot.UniformBytesUsed); + Assert.True(slot.TryAllocateUniforms(100, out var allocation)); + Assert.Equal(0UL, allocation.Offset % Math.Max(1UL, context.Capabilities.MinUniformBufferOffsetAlignment)); + Assert.Equal(0UL, allocation.Offset % Math.Max(1UL, context.Capabilities.MinStorageBufferOffsetAlignment)); + Assert.Equal(ring.UniformBuffer.Handle, allocation.Buffer.Handle); + allocations[index] = allocation; + stamps[index] = frame + 1; + *(int*)allocation.Pointer = stamps[index]; + for (int other = 0; other < 3; other++) + if (stamps[other] != 0) + { + Assert.Equal(stamps[other], *(int*)allocations[other].Pointer); + if (other != index) Assert.True(Math.Abs((long)allocation.Offset - allocations[other].Offset) >= 100); + } + Assert.False(slot.TryAllocateUniforms((int)slot.UniformCapacity + 1, out _)); + ulong serial = slot.RecordingSerial; + ulong partial = ring.SubmitPartial(); + Assert.Equal(partial, slot.LastSignalledValue); + Assert.True(slot.FrameValue > partial); + Assert.True(slot.RecordingSerial > serial); + Assert.True(slot.TryAllocateUniforms(100, out var next)); + Assert.True(next.Offset >= allocation.Offset + 100); + Assert.Equal(stamps[index], *(int*)allocation.Pointer); + ulong retiringAt = slot.FrameValue; + Parallel.For(0, 4, _ => ring.DeferDeletion(new Retired(() => { + Assert.True(ring.Timeline.FrameCompleted >= retiringAt); + retired++; + }))); + submitted[index] = ring.EndFrame(); + Assert.Equal(submitted[index], slot.LastSignalledValue); + } + ring.Timeline.WaitForFrame(ring.Timeline.FrameSignalled, WaitSite.DeviceWaitIdle); + ring.BeginFrame(); ring.EndFrame(); + Assert.Equal(18 * 4, retired); + Assert.Equal(0, ring.PendingDeletionCount); + } + ValidationAssert.NoErrors(messages); + } + + [SkippableFact] + public unsafe void DescriptorCacheReusesBindingsAndEvictsReleasedLifetimesAcrossPools() + { + var messages = new List(); + using (var context = OpenContext(messages)) + { + var binding = new DescriptorSetLayoutBinding(0, DescriptorType.CombinedImageSampler, 1, ShaderStageFlags.FragmentBit); + var layoutInfo = new DescriptorSetLayoutCreateInfo { + SType = StructureType.DescriptorSetLayoutCreateInfo, BindingCount = 1, PBindings = &binding, + }; + Assert.Equal(Result.Success, context.Api.CreateDescriptorSetLayout(context.Device, &layoutInfo, null, out var layout)); + var samplerInfo = new SamplerCreateInfo { + SType = StructureType.SamplerCreateInfo, MagFilter = Filter.Nearest, MinFilter = Filter.Nearest, + AddressModeU = SamplerAddressMode.ClampToEdge, AddressModeV = SamplerAddressMode.ClampToEdge, + AddressModeW = SamplerAddressMode.ClampToEdge, + }; + Sampler sampler = default; + try + { + Assert.Equal(Result.Success, context.Api.CreateSampler(context.Device, &samplerInfo, null, out sampler)); + using var image = new VulkanImage(context, 4, 4, Format.R8G8B8A8Unorm, ImageUsageFlags.SampledBit, ImageAspectFlags.ColorBit); + using var cache = new DescriptorCache(context); + DescriptorSetContents Contents(int program, ulong resource) => new(program, 0, + new[] { new SamplerBindingValue(0, image.View, sampler, resource) }, Array.Empty()); + var handles = new HashSet(); + for (int i = 1; i <= 700; i++) + { + var set = cache.Get(Contents(i, 10), layout); + Assert.True(handles.Add(set.Handle)); + Assert.Equal(set.Handle, cache.Get(Contents(i, 10), layout).Handle); + } + Assert.Equal(700, cache.Count); + Assert.Equal(700, cache.Hits); Assert.Equal(700, cache.Misses); + // Same Vulkan handles with a new lifetime must not hit a stale set. + var replacement = cache.Get(Contents(1, 11), layout); + Assert.DoesNotContain(replacement.Handle, handles); + cache.Release(10); + using (var released = cache.CollectReleases()) Assert.NotNull(released); + Assert.Equal(1, cache.Count); + Assert.Equal(replacement.Handle, cache.Get(Contents(1, 11), layout).Handle); + Assert.Null(cache.CollectReleases()); + cache.Release(11); + using (var released = cache.CollectReleases()) Assert.NotNull(released); + Assert.Equal(0, cache.Count); + } + finally + { + if (sampler.Handle != 0) context.Api.DestroySampler(context.Device, sampler, null); + context.Api.DestroyDescriptorSetLayout(context.Device, layout, null); + } + } + ValidationAssert.NoErrors(messages); + } + + private static int Target(VulkanDevice device, int image = 0) + { + if (image == 0) image = device.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int target = device.CreateFramebuffer(8, 8); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, image, 0); + device.SetDrawBuffers(target, 1); + return target; + } + + private static void Prepare(VulkanDevice device, int target, int program, int coverage) + { + device.BindFramebuffer(target); device.UseProgram(program); + device.SetViewport(0, 0, coverage, coverage); + device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); + device.SetColorMask(false, false, false, false); + } + + private static void Samples(VulkanDevice device, int query, int expected) + { + Assert.True(device.IsQueryResultAvailable(query)); + int actual = device.GetQueryResult(query); + if (device.PreciseOcclusionForTests) Assert.Equal(expected, actual); + else Assert.InRange(actual, 1, int.MaxValue - 1); + } + + [SkippableFact] + public void QueryResultsSurviveSlotReuseWithoutExtraSubmissionsOrBlocking() + { + var device = OpenDevice(); + try + { + int target = Target(device), program = GpuTest.LinkProgram(device, Triangle, White, "submission-query"); + int[] queries = Enumerable.Range(0, 40).Select(_ => device.CreateOcclusionQuery()).ToArray(); + device.BeginFrame(); device.Present(); + long[] waits = WaitCounts(); + long submits = VulkanStats.WaitCount(WaitSite.QueueSubmit); + int frames = 0; + for (int round = 0; round < 4; round++) + { + device.BeginFrame(); + for (int i = 0; i < queries.Length; i++) + { + Prepare(device, target, program, (i + round) % 8 + 1); + device.BeginOcclusionQuery(queries[i]); device.DrawFullscreenTriangle(); device.EndOcclusionQuery(queries[i]); + Assert.False(device.IsQueryResultAvailable(queries[i])); + } + device.Present(); frames++; + // f+2 reuses f's slot and must have collected its query results. + for (int age = 1; age <= 4; age++) + { + device.BeginFrame(); + if (age >= 2) + for (int i = 0; i < queries.Length; i++) Samples(device, queries[i], (int)Math.Pow((i + round) % 8 + 1, 2)); + device.Present(); frames++; + } + } + Assert.Equal(waits, WaitCounts()); + Assert.Equal(frames, VulkanStats.WaitCount(WaitSite.QueueSubmit) - submits); + Assert.InRange(device.OcclusionQueryPoolsForTests, 2, 4); + foreach (int query in queries) device.DeleteQuery(query); + Assert.False(device.IsQueryResultAvailable(queries[0])); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void QueryContinuesAcrossFramebufferChangesAndPartialReadback() + { + var device = OpenDevice(); + try + { + int targetA = Target(device), targetB = Target(device); + int program = GpuTest.LinkProgram(device, Triangle, White, "submission-query-split"); + int[] probes = Enumerable.Range(0, 31).Select(_ => device.CreateOcclusionQuery()).ToArray(); + int spanning = device.CreateOcclusionQuery(); + device.BeginFrame(); + foreach (int probe in probes) + { + Prepare(device, targetA, program, 1); + device.BeginOcclusionQuery(probe); device.DrawFullscreenTriangle(); device.EndOcclusionQuery(probe); + } + Prepare(device, targetA, program, 4); + device.BeginOcclusionQuery(spanning); device.DrawFullscreenTriangle(); + device.BindFramebuffer(targetB); device.DrawFullscreenTriangle(); + uint pixel = 0; + device.ReadDefaultFramebuffer(0, 0, 1, 1, (IntPtr)(&pixel)); + device.DrawFullscreenTriangle(); device.EndOcclusionQuery(spanning); + device.Present(); + for (int i = 0; i < 4; i++) { device.BeginFrame(); device.Present(); } + Samples(device, spanning, 48); + foreach (int probe in probes) Samples(device, probe, 1); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + private static byte[] Pixels(int size, byte red, byte green, byte blue) + { + var pixels = new byte[size * size * 4]; + for (int i = 0; i < pixels.Length; i += 4) + { pixels[i] = red; pixels[i + 1] = green; pixels[i + 2] = blue; pixels[i + 3] = 255; } + return pixels; + } + + private static unsafe void Upload(VulkanDevice device, int texture, byte[] pixels) + { + fixed (byte* pointer = pixels) + device.UploadTexture2D(texture, 0, 0, 0, 8, 8, EnumTexturePixelFormat.Rgba, (IntPtr)pointer); + } + + private static unsafe byte[] Read(VulkanDevice device, int target) + { + var pixels = new byte[8 * 8 * 4]; + device.BindFramebuffer(target); + fixed (byte* pointer = pixels) device.ReadDefaultFramebuffer(0, 0, 8, 8, (IntPtr)pointer); + return pixels; + } + + [SkippableFact] + public unsafe void WorkerUploadsAndGeneratedMipsReachTheNextFrameWithoutUploadWaits() + { + var device = OpenDevice(); + try + { + int[] textures = Enumerable.Range(0, 4).Select(_ => device.CreateTexture2D(8, 8, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false)).ToArray(); + int[] targets = textures.Select(texture => Target(device, texture)).ToArray(); + int renderTarget = Target(device), renderProgram = GpuTest.LinkProgram(device, Triangle, White, "upload-concurrent-render"); + int mip = device.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, true); + device.BeginFrame(); device.Present(); + long[] waits = WaitCounts(); + long blocking = VulkanStats.BlockingUploads, uploads = VulkanStats.UploadRequests; + for (int frame = 0; frame < 24; frame++) + { + int current = frame; + device.BeginFrame(); + var worker = Task.Run(() => { + for (int i = 0; i < textures.Length; i++) + Upload(device, textures[i], Pixels(8, (byte)(current * 9), (byte)(i * 41), 173)); + }); + try + { + Upload(device, mip, Pixels(8, (byte)(frame * 9), 71, 163)); + device.GenerateMipmaps(mip); + // Record rendering while a worker independently fills the upload batch. + Prepare(device, renderTarget, renderProgram, 8); device.SetColorMask(true, true, true, true); + device.DrawFullscreenTriangle(); + } + finally { worker.GetAwaiter().GetResult(); } + device.Present(); + } + Assert.Equal(waits, WaitCounts()); + Assert.Equal(blocking, VulkanStats.BlockingUploads); + Assert.True(VulkanStats.UploadRequests - uploads >= 24 * 6); + // The next submission carries any remaining batch before frame N+2 reads it. + device.BeginFrame(); device.Present(); + Assert.Equal(device.TimelineForTests.TransferRecorded, device.TimelineForTests.TransferSignalled); + device.BeginFrame(); + for (int i = 0; i < textures.Length; i++) + Assert.Equal(Pixels(8, 207, (byte)(i * 41), 173), Read(device, targets[i])); + Assert.Equal(Pixels(1, 207, 71, 163), device.ReadBackLevelForTests(mip, 3)); + Assert.Equal(Pixels(8, 255, 255, 255), Read(device, renderTarget)); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void ReuploadAndMipGenerationOnlyChangeDrawsRecordedAfterThem() + { + var device = OpenDevice(); + try + { + int source = device.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, true); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D source; + out vec4 color; + void main() { color = textureLod(source, vec2(0.5), 3.0); } + """, "submission-reupload"); + int sampler = device.CreateSampler(false); + int before = Target(device), after = Target(device), nextFrame = Target(device); + device.BeginFrame(); device.Present(); + long submits = VulkanStats.WaitCount(WaitSite.QueueSubmit), blocking = VulkanStats.BlockingUploads; + device.BeginFrame(); + Upload(device, source, Pixels(8, 255, 0, 0)); device.GenerateMipmaps(source); + Prepare(device, before, program, 8); device.SetColorMask(true, true, true, true); + device.SetSamplerUnit(program, "source", 0); device.BindTexture(0, source); device.BindSampler(0, sampler); + device.DrawFullscreenTriangle(); + Upload(device, source, Pixels(8, 0, 255, 0)); device.GenerateMipmaps(source); + device.BindFramebuffer(after); device.DrawFullscreenTriangle(); + Assert.Equal(submits, VulkanStats.WaitCount(WaitSite.QueueSubmit)); + Assert.Equal(Pixels(8, 255, 0, 0), Read(device, before)); + Assert.Equal(Pixels(8, 0, 255, 0), Read(device, after)); + device.Present(); + device.BeginFrame(); + Prepare(device, nextFrame, program, 8); device.SetColorMask(true, true, true, true); + device.BindTexture(0, source); device.BindSampler(0, sampler); device.DrawFullscreenTriangle(); + Assert.Equal(Pixels(8, 0, 255, 0), Read(device, nextFrame)); + device.Present(); + Assert.Equal(blocking, VulkanStats.BlockingUploads); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void OpenTransferBatchRetainsOverflowStagingUntilSubmissionCompletes() + { + var messages = new List(); + using (var context = OpenContext(messages)) + using (var ring = new FrameRing(context, framesInFlight: 2, uniformRingSize: 65536, stagingPerSlot: 65536)) + using (var textures = new TextureManager(context, ring.Uploads)) + { + var sizes = new uint[] { 8, 256 }; + var images = sizes.Select(size => textures.Create(size, size, Format.R8G8B8A8Unorm)).ToArray(); + var expected = sizes.Select(size => Enumerable.Range(0, (int)(size * size * 4)).Select(i => (byte)(i * 13 % 251)).ToArray()).ToArray(); + long overflows = VulkanStats.StagingOverflows; + for (int i = 0; i < images.Length; i++) + fixed (byte* pointer = expected[i]) textures.Upload(images[i], 0, 0, 0, sizes[i], sizes[i], (IntPtr)pointer, 4); + Assert.Equal(overflows + 1, VulkanStats.StagingOverflows); + ulong transfer = ring.Timeline.TransferRecorded; + Assert.True(transfer > ring.Timeline.TransferSignalled); + bool retired = false; + ring.DeferDeletion(new Retired(() => { + Assert.True(ring.Timeline.TransferCompleted >= transfer); + retired = true; + })); + ring.BeginFrame(); + Assert.False(retired); + Assert.True(ring.PendingDeletionCount >= 2); + ring.EndFrame(); + Assert.Equal(transfer, ring.Timeline.TransferSignalled); + ring.Timeline.WaitForFrame(ring.Timeline.FrameSignalled, WaitSite.Readback); + ring.BeginFrame(); + Assert.True(retired); Assert.Equal(0, ring.PendingDeletionCount); + ring.EndFrame(); + for (int i = 0; i < images.Length; i++) + { + using var readback = new VulkanBuffer(context, (ulong)expected[i].Length, BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + var commands = ring.Uploads.BeginRecording(inlineInFrame: false); + try + { + var image = textures.Get(images[i])!; + textures.TransitionTexture(commands, image, ImageLayout.TransferSrcOptimal); + var region = new BufferImageCopy { ImageSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + ImageExtent = new Extent3D(sizes[i], sizes[i], 1) }; + context.Api.CmdCopyImageToBuffer(commands, image.Image, ImageLayout.TransferSrcOptimal, readback.Handle, 1, ®ion); + } + finally { ring.Uploads.EndRecording(); } + ring.Timeline.WaitForTransfer(ring.Uploads.SubmitStandalone(), WaitSite.Readback); + Assert.True(new ReadOnlySpan((void*)readback.Mapped, expected[i].Length).SequenceEqual(expected[i])); + } + } + ValidationAssert.NoErrors(messages); + } + + [SkippableTheory] + [InlineData(true)] + [InlineData(false)] + public void DynamicStateReusePreservesPixelsAcrossPartialSubmissions(bool cached) + { + var device = OpenDevice(); + try + { + device.DynamicStateMaskingForTests = cached; + int target = Target(device); + int red = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + out vec4 color; + void main() { color = vec4(1, 0, 0, 1); } + """, "state-red"); + int green = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + out vec4 color; + void main() { color = vec4(0, 1, 0, 1); } + """, "state-green"); + int complete = device.DynamicStateCommandsPerDrawForTests; + for (int frame = 0; frame < 3; frame++) + { + device.BeginFrame(); + Prepare(device, target, red, 8); device.SetColorMask(true, true, true, true); + device.SetScissorEnabled(false); device.ClearColor(0, 0, 0, 0, 1); + device.SetViewport(0, 0, 4, 8); + long commands = device.DynamicStateCommandsForTests; + for (int i = 0; i < 32; i++) device.DrawFullscreenTriangle(); + Assert.Equal((cached ? 1 : 32) * complete, device.DynamicStateCommandsForTests - commands); + device.UseProgram(green); device.SetViewport(0, 0, 8, 8); + device.SetScissorEnabled(true); device.SetScissor(4, 4, 4, 4); + device.DrawFullscreenTriangle(); + byte[] pixels = Read(device, target); + for (int y = 0; y < 8; y++) + for (int x = 0; x < 8; x++) + { + int at = (y * 8 + x) * 4; + Assert.Equal(x < 4 ? 255 : 0, pixels[at]); + Assert.Equal(x >= 4 && y >= 4 ? 255 : 0, pixels[at + 1]); + Assert.Equal(0, pixels[at + 2]); Assert.Equal(255, pixels[at + 3]); + } + commands = device.DynamicStateCommandsForTests; + device.DrawFullscreenTriangle(); // Same state, new command buffer after readback. + Assert.Equal(complete, device.DynamicStateCommandsForTests - commands); + device.Present(); + } + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void IndirectOverflowAndLaterGrowthPreserveEveryDrawsRegion() + { + var device = OpenDevice(); + try + { + device.IndirectMinimumCapacityForTests = 40; // Two indexed commands fit initially. + var positions = new float[48]; var indices = new int[24]; + for (int strip = 0; strip < 4; strip++) + { + float left = -1 + strip * 0.5f, right = left + 0.5f; + new[] { left, -1f, 0f, right, -1f, 0f, right, 1f, 0f, left, 1f, 0f }.CopyTo(positions, strip * 12); + new[] { 0, 1, 2, 0, 2, 3 }.Select(index => index + strip * 4).ToArray().CopyTo(indices, strip * 6); + } + int mesh = device.CreateMesh(new MeshData(16, 24) { xyz = positions, Indices = indices, VerticesCount = 16, IndicesCount = 24 }, false); + int program = GpuTest.LinkProgram(device, """ + #version 330 core + layout(location = 0) in vec3 position; + void main() { gl_Position = vec4(position, 1); } + """, """ + #version 330 core + uniform float tint; + out vec4 color; + void main() { color = vec4(tint, 1, 1, 1); } + """, "indirect-region-lifetime"); + int tint = device.GetUniformLocation(program, "tint"); + int[] targets = Enumerable.Range(0, 3).Select(_ => Target(device)).ToArray(); + for (int frame = 0; frame < 3; frame++) + { + device.BeginFrame(); Prepare(device, targets[frame], program, 8); + device.SetColorMask(true, true, true, true); device.ClearColor(0, 0, 0, 0, 1); + device.SetUniform(program, tint, (30 + frame * 60) / 255f); + for (int strip = 0; strip < 4; strip++) + device.DrawMeshMulti(mesh, new[] { strip * 6 * sizeof(int), 0 }, new[] { 6 }, 1, false); + Assert.Equal(2, device.IndirectOverflowsForTests); + if (frame == 0) Assert.Equal(40UL, device.IndirectRingForTests.CapacityOf(device.CurrentSlotForTests)); + if (frame == 2) Assert.True(device.IndirectGrowthsForTests > 0); + device.Present(); + } + // Read only after slot reuse; early readbacks would hide lifetime mistakes. + device.BeginFrame(); + for (int frame = 0; frame < targets.Length; frame++) + Assert.Equal(Pixels(8, (byte)(30 + frame * 60), 255, 255), Read(device, targets[frame])); + device.Present(); device.DeleteMesh(mesh); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } +} diff --git a/Optimum.Render.Vulkan.Tests/Support/MotionFixture.cs b/Optimum.Render.Vulkan.Tests/Support/MotionFixture.cs new file mode 100644 index 00000000..ae191701 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Support/MotionFixture.cs @@ -0,0 +1,196 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Xunit; + +namespace Optimum.Render.Vulkan.Tests; + +/// Shared shader and uniform setup for the independent motion scenarios. +internal static class MotionFixture +{ + internal readonly record struct MotionTarget(int Framebuffer, int ColorTexture, int MotionTexture); + + internal static MotionTarget CreateMotionTarget(VulkanDevice device, int size) + { + int color = device.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int glow = device.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int motion = device.CreateTexture2D(size, size, EnumTextureInternalFormat.Rgba16f, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int depth = device.CreateTexture2D(size, size, EnumTextureInternalFormat.DepthComponent32, + EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + + int framebuffer = device.CreateFramebuffer(size, size); + device.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment0, color, 0); + device.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment1, glow, 0); + device.AttachTexture(framebuffer, EnumFramebufferAttachment.ColorAttachment2, motion, 0); + device.AttachTexture(framebuffer, EnumFramebufferAttachment.DepthAttachment, depth, 0); + device.SetDrawBuffers(framebuffer, 0b111); + Assert.True(device.CheckFramebufferComplete(framebuffer, out string status), status); + return new MotionTarget(framebuffer, color, motion); + } + + internal static float[] ReadMotion(VulkanDevice device, int texture, int size) + { + OptimumTextureReadback? readback = device.ReadTextureForParity(texture); + Assert.NotNull(readback); + Assert.Equal(size, readback.Width); + Assert.Equal(size, readback.Height); + Assert.NotNull(readback.Floats); + Assert.Equal(size * size * 4, readback.Floats.Length); + return readback.Floats; + } + + internal static MeshData CreateFaceData(float z = 0, int flags = 7 << 18) + { + var face = new MeshData(4, 6, withNormals: false, withUv: true, withRgba: true, withFlags: true); + float[] xy = [-0.5f, -0.5f, 0.5f, -0.5f, 0.5f, 0.5f, -0.5f, 0.5f]; + float[] uv = [0, 0, 1, 0, 1, 1, 0, 1]; + for (int i = 0; i < 4; i++) + face.AddVertexWithFlags(xy[i * 2], xy[i * 2 + 1], z, + uv[i * 2], uv[i * 2 + 1], Vintagestory.API.MathTools.ColorUtil.WhiteArgb, + flags); + foreach (int index in new[] { 0, 1, 2, 0, 2, 3 }) face.AddIndex(index); + return face; + } + + internal static int CreateFaceMesh(VulkanDevice device, float z = 0) + { + int mesh = device.CreateMesh(CreateFaceData(z), staticDraw: true); + Assert.True(mesh > 0, device.GetError() ?? "motion face upload failed"); + return mesh; + } + + internal static void SetFloat(VulkanDevice seam, int program, string name, float value) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, value); + } + + internal static void SetInt(VulkanDevice seam, int program, string name, int value) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, value); + } + + internal static void SetFloat2(VulkanDevice seam, int program, string name, float x, float y) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, x, y); + } + + internal static void SetFloat3(VulkanDevice seam, int program, string name, float x, float y, float z) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, x, y, z); + } + + internal static void SetFloat4( + VulkanDevice seam, int program, string name, float x, float y, float z, float w) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniform(program, location, x, y, z, w); + } + + internal static void SetMatrix(VulkanDevice seam, int program, string name, float[] matrix) + { + int location = seam.GetUniformLocation(program, name); + if (location >= 0) seam.SetUniformMatrix(program, location, matrix); + } + + internal static unsafe int CreateWhiteTexture(VulkanDevice seam) + { + var white = new byte[] { 255, 255, 255, 255 }; + fixed (byte* pixels = white) + { + return seam.CreateTexture2D(1, 1, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, (IntPtr)pixels, false); + } + } + + internal static int BindEveryDeclaredSampler( + VulkanDevice device, VulkanDevice seam, int programId) + { + int unit = 0; + foreach (string samplerName in device.SamplerNamesOf(programId)) + { + int texture = CreateWhiteTexture(seam); + seam.SetSamplerUnit(programId, samplerName, unit); + seam.BindTexture(unit, texture); + unit++; + } + return unit; + } + + internal static int LinkFromCorpus( + VulkanDevice seam, List stages, string name, bool oit) + { + var program = new CorpusProgram { PassName = name, Oit = oit }; + + foreach (ShaderStageSource stage in stages) + { + var shader = new CorpusShader + { + Type = stage.Stage, + Code = stage.Code, + PrefixCode = stage.PrefixCode ?? "", + }; + Assert.True(seam.CompileShader(shader), name + ": " + (seam.GetError() ?? "compile failed")); + + if (stage.Stage == EnumShaderType.VertexShader) program.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) program.FragmentShader = shader; + else program.GeometryShader = shader; + } + + int programId = seam.LinkProgram(program); + Assert.True(programId > 0, name + ": " + (seam.GetError() ?? "link failed")); + return programId; + } + + private sealed class CorpusShader : IShader + { + public EnumShaderType Type { get; set; } + public string Code { get; set; } = ""; + public string PrefixCode { get; set; } = ""; + public bool Compile() => true; + } + + private sealed class CorpusProgram : IShaderProgram + { + public int ProgramId { get; set; } + public string AssetDomain { get; set; } = "game"; + public int PassId { get; set; } + public string PassName { get; set; } = ""; + public bool ClampTexturesToEdge { get; set; } + public IShader VertexShader { get; set; } = null!; + public IShader FragmentShader { get; set; } = null!; + public IShader GeometryShader { get; set; } = null!; + public bool Oit { get; set; } + public bool Disposed => false; + public bool LoadError => false; + public Vintagestory.API.Datastructures.OrderedDictionary UBOs { get; } = new(); + public bool Compile() => true; + public bool HasUniform(string uniformName) => false; + public void Use() { } + public void Stop() { } + public void Dispose() { } + public void Uniform(string uniformName, float value) { } + public void Uniform(string uniformName, int value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2f value) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec2i value) { } + public void Uniform(string uniformName, float valueX, float valueY) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec3f value) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ) { } + public void Uniform(string uniformName, float valueX, float valueY, float valueZ, float valueW) { } + public void Uniform(string uniformName, Vintagestory.API.MathTools.Vec4f value) { } + public void Uniforms4(string uniformName, int count, float[] values) { } + public void UniformMatrix(string uniformName, float[] matrix) { } + public void BindTexture2D(string samplerName, int textureId, int textureNumber) { } + public void BindTextureCube(string samplerName, int textureId, int textureNumber) { } + public void UniformMatrices(string uniformName, int count, float[] matrix) { } + public void UniformMatrices4x3(string uniformName, int count, float[] matrix) { } + } +} diff --git a/Optimum.Render.Vulkan.Tests/Support/PipelineFixture.cs b/Optimum.Render.Vulkan.Tests/Support/PipelineFixture.cs new file mode 100644 index 00000000..aad4c2c0 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/Support/PipelineFixture.cs @@ -0,0 +1,534 @@ +// Source: Optimum.Render.Vulkan.Tests/PipelineKeyState.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +/// +/// The OpenGL-shaped state record the renderer used before every draw became native, kept for the +/// component tests that drive , +/// and directly: it builds their pipeline keys, blend sets and dynamic +/// state the way the removed emulated draw did. Nothing in the renderer uses it. +/// +internal sealed class PipelineKeyState +{ + public const int MaxColorAttachments = RenderLimits.MaxColorAttachments; + public const int MaxTextureUnits = RenderLimits.MaxTextureUnits; + + private readonly AttachmentBlend[] _blend = new AttachmentBlend[MaxColorAttachments]; + private readonly Interner _blendSignatures = new(); + private readonly Interner _targetFormats = new(); + + private int _cachedBlendId = -1; + private int _cachedBlendCount = -1; + private ColorComponentFlags _colorWriteMask = + ColorComponentFlags.RBit | ColorComponentFlags.GBit + | ColorComponentFlags.BBit | ColorComponentFlags.ABit; + + public PipelineKeyState() + { + for (int i = 0; i < _blend.Length; i++) _blend[i] = AttachmentBlend.Default; + } + + // ------------------------------------------------------------ dynamic state + + public Rect2D Viewport { get; private set; } + public Rect2D Scissor { get; private set; } + public bool ScissorEnabled { get; private set; } + + public bool DepthTest { get; private set; } + public bool DepthWrite { get; private set; } = true; + public CompareOp DepthCompare { get; private set; } = CompareOp.Less; + + public bool CullEnabled { get; private set; } + public CullModeFlags CullMode { get; private set; } = CullModeFlags.BackBit; + + public bool StencilTest { get; private set; } + public uint StencilWriteMask { get; private set; } = 0xFF; + public uint StencilCompareMask { get; private set; } = 0xFF; + public uint StencilReference { get; private set; } + public CompareOp StencilCompare { get; private set; } = CompareOp.Always; + public StencilOp StencilFail { get; private set; } = StencilOp.Keep; + public StencilOp StencilDepthFail { get; private set; } = StencilOp.Keep; + public StencilOp StencilPass { get; private set; } = StencilOp.Keep; + + public float LineWidth { get; private set; } = 1.0f; + public PrimitiveTopology Topology { get; private set; } = PrimitiveTopology.TriangleList; + + // ------------------------------------------------------------- pipeline state + + public PolygonMode PolygonMode { get; private set; } = PolygonMode.Fill; + public int CurrentProgram { get; private set; } + + public const FrontFace FrontFace = RenderLimits.FrontFace; + + // -------------------------------------------------------------------- setters + + public void SetViewport(int x, int y, int width, int height) => + Viewport = new Rect2D(new Offset2D(x, y), new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + + /// + /// Records the scissor rectangle, clipped to the positive quadrant. + /// + /// glScissor takes a signed origin and the game passes negative ones - a + /// dialog that extends past the top of the screen produces y = -72. GL clips + /// the rectangle against the framebuffer and keeps the visible remainder; + /// Vulkan rejects a negative offset outright. Moving the origin back to zero + /// and taking the same amount off the extent leaves the identical region. + /// + public void SetScissor(int x, int y, int width, int height) + { + int clippedX = Math.Max(0, x); + int clippedY = Math.Max(0, y); + width -= clippedX - x; + height -= clippedY - y; + + Scissor = new Rect2D( + new Offset2D(clippedX, clippedY), + new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + } + + public void SetScissorEnabled(bool enabled) => ScissorEnabled = enabled; + + public void SetDepthTest(bool enabled) => DepthTest = enabled; + public void SetDepthWrite(bool enabled) => DepthWrite = enabled; + public void SetDepthFunc(int glFunc) => DepthCompare = GlEnums.CompareOpFrom(glFunc); + + public void SetCullEnabled(bool enabled) => CullEnabled = enabled; + public void SetCullBack(bool back) => CullMode = back ? CullModeFlags.BackBit : CullModeFlags.FrontBit; + + public void SetStencilTest(bool enabled) => StencilTest = enabled; + public void SetStencilMask(int mask) => StencilWriteMask = (uint)mask; + + public void SetStencilFunc(int func, int reference, int mask) + { + StencilCompare = GlEnums.CompareOpFrom(func); + StencilReference = (uint)reference; + StencilCompareMask = (uint)mask; + } + + public void SetStencilOp(int fail, int depthFail, int pass) + { + StencilFail = GlEnums.StencilOpFrom(fail); + StencilDepthFail = GlEnums.StencilOpFrom(depthFail); + StencilPass = GlEnums.StencilOpFrom(pass); + } + + public void SetLineWidth(float width) => LineWidth = width; + public void SetTopology(EnumDrawMode mode) => Topology = GlEnums.TopologyFrom(mode); + public void SetWireframe(bool enabled) => PolygonMode = enabled ? PolygonMode.Line : PolygonMode.Fill; + public void SetProgram(int programId) => CurrentProgram = programId; + + /// + /// GL's colour mask is global; Vulkan's is per attachment. Setting it here + /// replicates it across all of them, which is what the GL behaviour means. + /// + public void SetColorMask(bool r, bool g, bool b, bool a) + { + ColorComponentFlags mask = 0; + if (r) mask |= ColorComponentFlags.RBit; + if (g) mask |= ColorComponentFlags.GBit; + if (b) mask |= ColorComponentFlags.BBit; + if (a) mask |= ColorComponentFlags.ABit; + + if (mask == _colorWriteMask) return; + _colorWriteMask = mask; + + for (int i = 0; i < _blend.Length; i++) _blend[i].WriteMask = mask; + InvalidateBlend(); + } + + /// + /// Applies one of the game's named blend modes to every attachment, matching + /// the factor pairs ClientPlatformWindows.GlToggleBlend selects. + /// + public void SetBlend(bool enabled, EnumBlendMode mode) + { + (BlendFactor srcColor, BlendFactor dstColor, BlendFactor srcAlpha, BlendFactor dstAlpha) = + AttachmentBlend.FactorsFor(mode); + + for (int i = 0; i < _blend.Length; i++) + { + _blend[i].Enabled = enabled; + _blend[i].SrcColor = srcColor; + _blend[i].DstColor = dstColor; + _blend[i].ColorOp = BlendOp.Add; + _blend[i].SrcAlpha = srcAlpha; + _blend[i].DstAlpha = dstAlpha; + _blend[i].AlphaOp = BlendOp.Add; + } + InvalidateBlend(); + } + + /// glEnable/glDisable(GL_BLEND) preserve the indexed blend functions. + public void SetBlendEnabled(bool enabled) + { + for (int i = 0; i < _blend.Length; i++) _blend[i].Enabled = enabled; + InvalidateBlend(); + } + + /// + /// Per-attachment blend, which the OIT and SSAO passes use through + /// glBlendFunci and glBlendEquationi. + /// + public void SetAttachmentBlendFunc(int attachment, int srcColor, int dstColor, int srcAlpha, int dstAlpha) + { + if ((uint)attachment >= MaxColorAttachments) return; + + _blend[attachment].SrcColor = GlEnums.BlendFactorFrom(srcColor); + _blend[attachment].DstColor = GlEnums.BlendFactorFrom(dstColor); + _blend[attachment].SrcAlpha = GlEnums.BlendFactorFrom(srcAlpha); + _blend[attachment].DstAlpha = GlEnums.BlendFactorFrom(dstAlpha); + InvalidateBlend(); + } + + public void SetAttachmentBlendEquation(int attachment, int equation) + { + if ((uint)attachment >= MaxColorAttachments) return; + + BlendOp op = GlEnums.BlendOpFrom(equation); + _blend[attachment].ColorOp = op; + _blend[attachment].AlphaOp = op; + InvalidateBlend(); + } + + // ---------------------------------------------------------------------- keys + + /// + /// Interns the current blend set. The id is cached and only recomputed after + /// a blend change, so a run of draws sharing state pays nothing. + /// + public int BlendId(int attachmentCount) + { + int count = Math.Clamp(attachmentCount, 0, MaxColorAttachments); + if (_cachedBlendId >= 0 && _cachedBlendCount == count) return _cachedBlendId; + + _cachedBlendCount = count; + _cachedBlendId = _blendSignatures.Intern(new BlendSignature(_blend.AsSpan(0, count))); + return _cachedBlendId; + } + + public int InternTargetFormats(RenderTargetFormats formats) => _targetFormats.Intern(formats); + public RenderTargetFormats TargetFormats(int id) => _targetFormats.Get(id); + + /// Blend state for one attachment, for pipeline creation. + public AttachmentBlend BlendFor(int attachment) => _blend[attachment]; + + public PipelineKey BuildKey(int vertexLayoutId, int targetFormatsId, int attachmentCount) => + BuildKey(vertexLayoutId, targetFormatsId, attachmentCount, uint.MaxValue); + + /// The key under the current for a target with these draw buffers. + public PipelineKey BuildKey(int vertexLayoutId, int targetFormatsId, int attachmentCount, uint drawBufferMask) => new( + ProgramId: CurrentProgram, + VertexLayoutId: vertexLayoutId, + TargetFormatsId: targetFormatsId, + BlendId: PipelineBlendId(attachmentCount, drawBufferMask), + PolygonMode: PolygonMode, + TopologyClass: GlEnums.TopologyClassOf(Topology)); + + // ------------------------------------------------------- colour write masks + + private const ColorComponentFlags AllChannels = + ColorComponentFlags.RBit | ColorComponentFlags.GBit | ColorComponentFlags.BBit | ColorComponentFlags.ABit; + + private int _cachedPipelineBlendId = -1; + private int _cachedPipelineBlendCount = -1; + private uint _cachedPipelineDrawBuffers; + + /// + /// How draw-buffer and colour-mask changes reach the GPU (Phase 2, C4). The + /// device sets it from the context's selected tier; component tests keep the + /// default, which bakes everything into the pipeline key. + /// + public ColorWriteTier ColorWriteTier + { + get => _colorWriteTier; + set + { + _colorWriteTier = value; + InvalidateBlend(); + } + } + + private ColorWriteTier _colorWriteTier = ColorWriteTier.PipelineKey; + private bool _dynamicBlend; + + /// With the mask tier: blend enable and equation are dynamic too, so the blend set leaves the key. + public bool DynamicBlend + { + get => _dynamicBlend; + set + { + _dynamicBlend = value; + InvalidateBlend(); + } + } + + /// The global glColorMask. + public ColorComponentFlags ColorMask => _colorWriteMask; + + private void InvalidateBlend() + { + _cachedBlendId = -1; + _cachedBlendCount = -1; + _cachedPipelineBlendId = -1; + _cachedPipelineBlendCount = -1; + } + + /// Bit i set when the program statically writes fragment output i. + public static uint OutputBits(HashSet writtenOutputs) + { + uint bits = 0; + for (int i = 0; i < MaxColorAttachments; i++) + { + if (writtenOutputs.Contains(i)) bits |= 1u << i; + } + return bits; + } + + /// + /// The effective write mask of one attachment: drawBufferEnabled ? colorMask : 0, + /// then masked by the outputs the program writes (an unwritten output keeps + /// the attachment's contents, as GL does; Vulkan would store undefined values). + /// + public ColorComponentFlags EffectiveWriteMask(int attachment, uint drawBufferMask, uint writtenOutputs) + { + if ((uint)attachment >= MaxColorAttachments) return 0; + if (((drawBufferMask >> attachment) & 1) == 0) return 0; + if (((writtenOutputs >> attachment) & 1) == 0) return 0; + return _colorWriteMask; + } + + /// + /// The blend state of one attachment as the pipeline bakes it under the tier: + /// the draw-buffer-masked write mask in the key tier, glColorMask alone in the + /// enable tier (draw buffers are the dynamic enable), and a canonical mask in + /// the mask tier (the dynamic mask replaces it; with dynamic blend the whole + /// attachment state is canonical). + /// + public AttachmentBlend PipelineBlendFor(int attachment, uint drawBufferMask) + { + AttachmentBlend blend = _blend[attachment]; + switch (ColorWriteTier) + { + case ColorWriteTier.PipelineKey: + if (((drawBufferMask >> attachment) & 1) == 0) blend.WriteMask = 0; + break; + case ColorWriteTier.DynamicMask: + if (DynamicBlend) blend = AttachmentBlend.Default; + blend.WriteMask = AllChannels; + break; + } + return blend; + } + + /// + /// The interned blend set a pipeline is keyed on under the tier. Equal to + /// whenever the tier leaves the state unchanged, so the + /// key tier with every draw buffer selected keys exactly as before. + /// + public int PipelineBlendId(int attachmentCount, uint drawBufferMask) + { + int count = Math.Clamp(attachmentCount, 0, MaxColorAttachments); + uint selectable = count == 32 ? uint.MaxValue : (1u << count) - 1; + uint relevant = drawBufferMask & selectable; + + if (ColorWriteTier == ColorWriteTier.DynamicEnable || + (ColorWriteTier == ColorWriteTier.PipelineKey && relevant == selectable)) + { + return BlendId(count); + } + + if (ColorWriteTier == ColorWriteTier.DynamicMask) relevant = 0; + if (_cachedPipelineBlendId >= 0 && _cachedPipelineBlendCount == count && _cachedPipelineDrawBuffers == relevant) + { + return _cachedPipelineBlendId; + } + + Span baked = stackalloc AttachmentBlend[count]; + for (int i = 0; i < count; i++) baked[i] = PipelineBlendFor(i, drawBufferMask); + + _cachedPipelineBlendCount = count; + _cachedPipelineDrawBuffers = relevant; + _cachedPipelineBlendId = _blendSignatures.Intern(new BlendSignature(baked)); + return _cachedPipelineBlendId; + } + + /// + /// Restores the defaults a fresh GL context would have. Called when the + /// device is created and whenever the client resets its own state wholesale. + /// + public void Reset() + { + for (int i = 0; i < _blend.Length; i++) _blend[i] = AttachmentBlend.Default; + InvalidateBlend(); + _colorWriteMask = ColorComponentFlags.RBit | ColorComponentFlags.GBit + | ColorComponentFlags.BBit | ColorComponentFlags.ABit; + + DepthTest = false; + DepthWrite = true; + DepthCompare = CompareOp.Less; + CullEnabled = false; + CullMode = CullModeFlags.BackBit; + ScissorEnabled = false; + StencilTest = false; + StencilWriteMask = 0xFF; + StencilCompareMask = 0xFF; + StencilReference = 0; + StencilCompare = CompareOp.Always; + StencilFail = StencilDepthFail = StencilPass = StencilOp.Keep; + LineWidth = 1.0f; + Topology = PrimitiveTopology.TriangleList; + PolygonMode = PolygonMode.Fill; + CurrentProgram = 0; + } +} +} + +// Source: Optimum.Render.Vulkan.Tests/SharedLayoutTestBinding.cs +namespace Optimum.Render.Vulkan.Tests +{ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Xunit; + +/// +/// Binds a program built outside a device the way the device's draw path does under +/// the shared pipeline layout (plan decision 9): each sampler's texture gets a slot in +/// a bindless table of its own and the index goes into the push block, set 1 is the +/// table's set, and set 2 carries the record buffer at its dynamic binding, a storage +/// buffer at every storage block's binding and a zero-filled placeholder everywhere else. +/// For component tests that draw through with a +/// command buffer of their own. +/// +internal sealed unsafe class SharedLayoutTestBinding : IDisposable +{ + private sealed class StillClock : ITimelineClock + { + public ulong FrameRecorded => 1; + public ulong TransferRecorded => 1; + public ulong FrameCompleted => 0; + public ulong TransferCompleted => 0; + } + + /// What one sampler reads. + public readonly record struct SampledTexture(int TextureId, SamplerState State, + ImageLayout Layout = ImageLayout.ShaderReadOnlyOptimal); + + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly DescriptorCache _descriptors; + private readonly VulkanBuffer _placeholder; + + public BindlessTextureTable Table { get; } + + public SharedLayoutTestBinding(VulkanContext context, TextureManager textures) + { + _context = context; + _textures = textures; + Table = new BindlessTextureTable(context, textures, new StillClock()); + _descriptors = new DescriptorCache(context); + _placeholder = new VulkanBuffer(context, 64 * 1024, + BufferUsageFlags.UniformBufferBit | BufferUsageFlags.StorageBufferBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + new Span((void*)_placeholder.Mapped, 64 * 1024).Clear(); + } + + /// + /// Before the rendering scope opens: the table's placeholders and every texture in + /// that is read shader-read-only go to that layout. + /// + public void Transition(CommandBuffer commandBuffer, IEnumerable sampled) + { + for (int kind = 0; kind < BindlessKinds.Count; kind++) + { + VulkanTexture placeholder = _textures.Get(Table.PlaceholderTextureId((TextureKind)kind))!; + _textures.TransitionTexture(commandBuffer, placeholder, ImageLayout.ShaderReadOnlyOptimal); + } + foreach (SampledTexture texture in sampled) + { + if (texture.Layout != ImageLayout.ShaderReadOnlyOptimal) continue; + _textures.TransitionTexture(commandBuffer, _textures.Get(texture.TextureId)!, ImageLayout.ShaderReadOnlyOptimal); + } + } + + /// + /// Binds sets 1 and 2 and pushes the slot indices for one draw of + /// through its own pipeline layout. Every sampler the + /// program declares must be in ; frame textures (set 0) + /// are not supported here. + /// + public void Bind(CommandBuffer commandBuffer, ShaderProgramResources program, + IReadOnlyDictionary samplers, VulkanBuffer? record = null, VulkanBuffer? storage = null) + { + Vk api = _context.Api; + PipelineLayout layout = program.PipelineLayout; + SharedPipelineLayout shape = program.StandaloneLayout + ?? throw new InvalidOperationException("the program was built by a device; bind through the device"); + + int pushSize = program.Interface.PushConstantSize; + if (pushSize > 0) + { + var push = new byte[pushSize]; + foreach (SamplerBinding sampler in program.Interface.Samplers) + { + Assert.False(sampler.IsFrameTexture, "frame textures are not bound by this helper: " + sampler.Name); + SampledTexture sampled = samplers[sampler.Name]; + uint slot = Table.Resolve(_textures.Get(sampled.TextureId), sampler.Kind, sampled.State, sampled.Layout); + Assert.NotEqual(0u, slot); + BitConverter.TryWriteBytes(push.AsSpan(sampler.PushOffset, ProgramInterfaceLayout.SlotBytes), slot); + } + Table.Flush(); + + DescriptorSet textureSet = Table.Set; + api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Graphics, layout, + (uint)SetConvention.TextureSet, 1, &textureSet, 0, null); + fixed (byte* bytes = push) + { + api.CmdPushConstants(commandBuffer, layout, SharedPipelineLayout.Stages, 0, (uint)pushSize, bytes); + } + } + + if (!program.Interface.UsesStorageSet) return; + + var buffers = new BufferBindingValue[SetConvention.StorageSetBindingCount]; + for (int binding = 0; binding < buffers.Length; binding++) + { + buffers[binding] = new BufferBindingValue((uint)binding, _placeholder.Handle, 0, _placeholder.Size, _placeholder.Id); + } + if (record != null) + { + buffers[SetConvention.ProgramRecordBinding] = new BufferBindingValue(SetConvention.ProgramRecordBinding, + record.Handle, 0, (ulong)Math.Max(program.UniformShadow.Length, 4), record.Id); + } + if (storage != null) + { + foreach (BlockBinding block in program.Interface.StorageBlocks) + { + buffers[block.Binding] = new BufferBindingValue((uint)block.Binding, storage.Handle, 0, storage.Size, storage.Id); + } + } + + DescriptorSet storageSet = _descriptors.Get( + new DescriptorSetContents(0, SetConvention.StorageSet, Array.Empty(), buffers), + shape.StorageSetLayout); + uint recordOffset = 0; + api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Graphics, layout, + (uint)SetConvention.StorageSet, 1, &storageSet, 1, &recordOffset); + } + + public void Dispose() + { + _context.Api.DeviceWaitIdle(_context.Device); + _descriptors.Dispose(); + _placeholder.Dispose(); + Table.Dispose(); + } +} +} diff --git a/Optimum.Render.Vulkan.Tests/TaaResolveTests.cs b/Optimum.Render.Vulkan.Tests/TaaResolveTests.cs new file mode 100644 index 00000000..1f0f138b --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/TaaResolveTests.cs @@ -0,0 +1,1691 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// Drives the real taa-resolve shader pair (see docs/vulkan.md) on the +/// Vulkan backend with synthetic inputs, the way +/// and drive the world programs: load through +/// , build a real pipeline against a three-attachment +/// MRT framebuffer (colour history RGBA16F, aux/glow RGBA8, linear depth R32F), +/// draw the fullscreen triangle and read back inside the frame. +/// +/// This goes one level lower than the device (VulkanDevice): the +/// public seam's EnumTextureInternalFormat has no R32F, and +/// ReadDefaultFramebuffer always assumes 4 bytes per pixel, neither of +/// which fits an HDR history or a float depth target. So this talks to +/// , and +/// directly and builds descriptor sets by +/// hand with a private , mirroring what +/// VulkanDevice does per draw but scoped to one fullscreen pass with named +/// uniforms and named samplers. +/// +public class TaaResolveTests +{ + private readonly ITestOutputHelper _output; + + public TaaResolveTests(ITestOutputHelper output) => _output = output; + + private const uint Size = 32; + + private static readonly float[] Identity4 = + { + 1f, 0f, 0f, 0f, + 0f, 1f, 0f, 0f, + 0f, 0f, 1f, 0f, + 0f, 0f, 0f, 1f, + }; + + private static bool TryCreateContext( + ITestOutputHelper output, List messages, out VulkanContext? context) => + GpuTest.TryCreateContext(output, messages, out context); + + // ------------------------------------------------------------------ tests + + /// + /// With resetHistory=1 the resolve must ignore whatever the history + /// holds entirely: the output is the current frame's colour, unmodified. + /// + [SkippableFact] + public unsafe void ResetHistoryIgnoresTheHistoryEntirely() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const float currentR = 0.7f, currentG = 0.3f, currentB = 0.2f; + var inputs = CreateInputSet(textures); + UploadFlatRgba16F(textures, inputs.SceneTex, currentR, currentG, currentB, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + // Motion says "written, no displacement" so the resolve does not + // fall back to a camera-reprojection path that would need a valid + // matrix; irrelevant here since reset overrides alpha regardless. + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + // History: a completely different colour. If this leaks into the + // output at all, reset is not doing its job. + UploadFlatRgba16F(textures, inputs.HistoryColor, 0.1f, 0.9f, 0.1f, 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + + var uniforms = new TaaUniforms { ResetHistory = 1 }; + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + Assert.InRange(ReadHalf(colorBytes, x, y, 0, 8), currentR - 0.02f, currentR + 0.02f); + Assert.InRange(ReadHalf(colorBytes, x, y, 1, 8), currentG - 0.02f, currentG + 0.02f); + Assert.InRange(ReadHalf(colorBytes, x, y, 2, 8), currentB - 0.02f, currentB + 0.02f); + } + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// A perfectly static scene (identity camera, zero jitter, zero motion) + /// converges to the constant current colour as the two history sets are + /// ping-ponged across many resolves. + /// + [SkippableFact] + public unsafe void StaticSceneConvergesToTheCurrentColour() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const float currentValue = 0.6f; + var inputs = CreateInputSet(textures); + UploadFlatRgba16F(textures, inputs.SceneTex, currentValue, currentValue, currentValue, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + + TaaAttachmentSet setA = CreateAttachmentSet(textures, targets); + TaaAttachmentSet setB = CreateAttachmentSet(textures, targets); + // Seed A far from the current colour so twenty resolves are a real + // convergence, not a no-op. + UploadFlatRgba16F(textures, setA.Color, 0f, 0f, 0f, 1f); + UploadFlatRgba8(textures, setA.Glow, 0, 0, 0, 255); + UploadFlatR32F(textures, setA.Depth, 0.5f); + + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.1f }; + + TaaAttachmentSet history = setA, current = setB; + for (int i = 0; i < 20; i++) + { + inputs.HistoryColor = history.Color; + inputs.HistoryGlow = history.Glow; + inputs.HistoryDepth = history.Depth; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, current); + + (history, current) = (current, history); + } + + // 'history' now holds the last write, since the pair swapped once + // more than it resolved. + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, history.Color, 8); + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + Assert.InRange(ReadHalf(colorBytes, x, y, 0, 8), currentValue - 0.02f, currentValue + 0.02f); + Assert.InRange(ReadHalf(colorBytes, x, y, 1, 8), currentValue - 0.02f, currentValue + 0.02f); + Assert.InRange(ReadHalf(colorBytes, x, y, 2, 8), currentValue - 0.02f, currentValue + 0.02f); + } + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// A uniform +2px motion field reprojects the history two pixels: a bright + /// band in the history shows up two columns earlier in the output. + /// + /// The current frame cannot be perfectly flat here. The resolve rectifies + /// history against the current frame's own 3x3 neighbourhood before + /// blending it in (see clipToBox in taa-resolve.fsh) - against a + /// genuinely flat current, that neighbourhood box has zero width and any + /// history value that disagrees with it, however it got there, is clipped + /// back to (effectively) the current colour. That is the clip working as + /// designed, not a test bug, so the current frame here carries a fine + /// per-column checker instead of a flat fill: it keeps the local box open + /// (both a low and a high value are always present in every 3x3 window) + /// without giving the resolve any large-scale feature of its own, so a + /// reprojected history feature is what a regional average actually shows. + /// + [SkippableFact] + public unsafe void UniformMotionReprojectsTheHistoryByThatOffset() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + var inputs = CreateInputSet(textures); + // Per-column checker: every 3-wide window has both 0.3 and 0.7, so + // clipToBox never degenerates to a point, but no column carries a + // feature of its own for the assertions to confuse with history's. + UploadRgba16F(textures, inputs.SceneTex, (x, _) => (x % 2 == 0) ? 0.3f : 0.7f, + (x, _) => (x % 2 == 0) ? 0.3f : 0.7f, (x, _) => (x % 2 == 0) ? 0.3f : 0.7f, (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + // Uniform +2px motion, written (matches depth) everywhere. + UploadFlatRgba16F(textures, inputs.MotionTex, 2f, 0f, 0f, 0.5f); + + const int stripeStart = 14, stripeWidth = 4; + const float background = 0.5f, stripe = 1.0f; + UploadRgba16F(textures, inputs.HistoryColor, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, _) => 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + // Mostly history, so the reprojected band dominates the blend. + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.05f }; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + + // Reading history at (pixelCentre + mv) means an output column c + // sees history column c + 2: the bright band at history columns + // [14,18) should appear at output columns [12,16). + float expectedBandAverage = AverageRed(colorBytes, stripeStart - 2, stripeStart - 2 + stripeWidth); + // A control window far from both the source and destination bands, + // still reading flat history background wherever it samples. + float controlAverage = AverageRed(colorBytes, 24, 28); + + _output.WriteLine($"reprojected band average={expectedBandAverage}, control average={controlAverage}"); + Assert.True(expectedBandAverage > controlAverage + 0.1f, + $"expected the reprojected band (avg {expectedBandAverage}) to read clearly brighter " + + $"than the control window (avg {controlAverage})"); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// A history pixel far outside the current frame's neighbourhood range - + /// a stale ghost, a lighting spike - is clipped back toward that + /// neighbourhood rather than blended in at full strength. + /// + [SkippableFact] + public unsafe void AnOutlierHistoryValueIsClippedTowardTheNeighbourhood() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const float currentValue = 0.5f; + var inputs = CreateInputSet(textures); + UploadFlatRgba16F(textures, inputs.SceneTex, currentValue, currentValue, currentValue, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + + // A magenta outlier, nothing like the current frame's flat grey. + const float outlierR = 1.0f, outlierG = 0.0f, outlierB = 1.0f; + UploadFlatRgba16F(textures, inputs.HistoryColor, outlierR, outlierG, outlierB, 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + // Even with heavy history weight, the clip should dominate. + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.5f }; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + float r = ReadHalf(colorBytes, (int)Size / 2, (int)Size / 2, 0, 8); + float g = ReadHalf(colorBytes, (int)Size / 2, (int)Size / 2, 1, 8); + float b = ReadHalf(colorBytes, (int)Size / 2, (int)Size / 2, 2, 8); + _output.WriteLine($"resolved=({r},{g},{b}) current=({currentValue}) outlier=({outlierR},{outlierG},{outlierB})"); + + Assert.InRange(r, currentValue - 0.1f, currentValue + 0.1f); + Assert.InRange(g, currentValue - 0.1f, currentValue + 0.1f); + Assert.InRange(b, currentValue - 0.1f, currentValue + 0.1f); + // Clearly not the raw outlier, in at least the channel it disagrees + // with the current frame the most. + Assert.True(MathF.Abs(g - outlierG) > 0.3f, "the outlier's green channel should have been clipped away"); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// Sky (depth == 1, nothing wrote the motion attachment) is a direction: a + /// camera translation must not move it. With a real perspective (far 60) + /// and the previous camera 8 blocks to the side, the finite reprojection + /// would slide the history band by 8 / 60 * 32 / tan(35 deg) = 6 px; the + /// infinite-direction path keeps it exactly where it is. The view has the + /// eye 1.7 blocks above the origin, as CameraMatrixOrigin does, so the + /// direction has to be far minus near rather than the far point alone. + /// + [SkippableFact] + public unsafe void SkyDoesNotMoveUnderCameraTranslation() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + const double near = 0.0689, far = 60.0, fov = 70.0 * Math.PI / 180.0, eyeHeight = 1.7; + double[] projection = Vintagestory.API.MathTools.Mat4d.Perspective(Vintagestory.API.MathTools.Mat4d.Create(), fov, 1.0, near, far); + double[] view = Vintagestory.API.MathTools.Mat4d.Identity(Vintagestory.API.MathTools.Mat4d.Create()); + view = Vintagestory.API.MathTools.Mat4d.Translate(view, view, 0.0, -eyeHeight, 0.0); + double[] viewProj = Vintagestory.API.MathTools.Mat4d.Mul(Vintagestory.API.MathTools.Mat4d.Create(), projection, view); + double[] inverse = Vintagestory.API.MathTools.Mat4d.Invert(Vintagestory.API.MathTools.Mat4d.Create(), viewProj)!; + float[] invF = Array.ConvertAll(inverse, v => (float)v); + float[] vpF = Array.ConvertAll(viewProj, v => (float)v); + float[] viewF = Array.ConvertAll(view, v => (float)v); + const float cameraDeltaX = 8f; + double finiteShift = cameraDeltaX / far * (Size / 2.0) / Math.Tan(fov / 2.0); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + var inputs = CreateInputSet(textures); + UploadRgba16F(textures, inputs.SceneTex, (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, + (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + // Sky everywhere, nothing wrote motion (a = 0). + UploadFlatR32F(textures, inputs.DepthTex, 1.0f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0f); + + const int stripeStart = 14, stripeWidth = 4; + const float background = 0.5f, stripe = 1.0f; + UploadRgba16F(textures, inputs.HistoryColor, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, _) => 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, (float)far); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + var uniforms = new TaaUniforms + { + ResetHistory = 0, + BlendAlpha = 0.05f, + InvViewProjJittered = invF, + PrevViewProj = vpF, + ViewMatrix = viewF, + CameraDelta = new[] { cameraDeltaX, 0f, 0f }, + }; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + double centroid = RedCentroidX(colorBytes, stripeStart - 8, stripeStart + stripeWidth + 8, background); + _output.WriteLine($"stripe centroid x = {centroid:F3}, expected {stripeStart + stripeWidth / 2.0:F1}; a finite reprojection would have moved it by {finiteShift:F1} px"); + Assert.True(finiteShift > 3.0, "the translation chosen is too small to tell the two paths apart"); + Assert.InRange(centroid, stripeStart + stripeWidth / 2.0 - 0.35, stripeStart + stripeWidth / 2.0 + 0.35); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// With non-zero jitter, the per-pixel Blackman-Harris reconstruction in + /// taa-resolve.fsh (the filtered/filteredWeight loop) is + /// supposed to undo the raster displacement: a scene that was rendered + /// with jitterPx=(+0.5,0) - meaning its texel at raster column x holds + /// the unjittered scene's value at (x + 0.5) - jitterPx, per the file's + /// own convention comment - should reconstruct to the same edge position + /// as a scene rendered with zero jitter that already holds the unjittered + /// values directly. Both runs use resetHistory=1 so the output is exactly + /// current, isolating the reconstruction step from history/blend. + /// + /// The edge is encoded as one-pixel-wide linear coverage ramp (not a hard + /// step) so a subpixel shift is representable in a texel grid at all; + /// the crossing point is then recovered from the *output* by a linear + /// interpolation against the 0.5 threshold - a "column-average centroid" + /// since the scene is flat in y. + /// + [SkippableFact] + public unsafe void JitteredReconstructionMatchesTheUnjitteredStaticEdge() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const float edgeCentre = 16f; + // One-pixel-wide linear coverage ramp around `pos`, standing in + // for a hard edge at edgeCentre that a discrete texel grid can + // still represent a subpixel shift of. + float EdgeAt(float pos) => Math.Clamp(pos - edgeCentre + 0.5f, 0f, 1f); + + float centroidJitterZero = ResolveEdgeCentroid( + context!, commands, textures, state, targets, pipelines, program, descriptors, + jitterPx: (0f, 0f), + // Baseline: texel at column x already holds the unjittered + // value at its own pixel centre. + sceneAt: x => EdgeAt(x + 0.5f)); + + float centroidJitteredHalfPx = ResolveEdgeCentroid( + context!, commands, textures, state, targets, pipelines, program, descriptors, + jitterPx: (0.5f, 0f), + // Jittered render: texel at column x holds the unjittered + // value at (x + 0.5) - jitterPx, per the file's convention. + sceneAt: x => EdgeAt(x + 0.5f - 0.5f)); + + _output.WriteLine($"centroid jitter=0: {centroidJitterZero}, centroid jitter=+0.5px: {centroidJitteredHalfPx}"); + Assert.True(Math.Abs(centroidJitteredHalfPx - centroidJitterZero) < 0.25f, + $"reconstructed edge moved by {Math.Abs(centroidJitteredHalfPx - centroidJitterZero)}px " + + "with the jitter; it should not move at all"); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// One resolve, reading back the reconstructed edge's crossing column + /// (0.5-threshold linear interpolation across the column averages) for + /// . + /// + private static unsafe float ResolveEdgeCentroid( + VulkanContext context, SetupQueue commands, TextureManager textures, PipelineKeyState state, + RenderTargetManager targets, GraphicsPipelineCache pipelines, ShaderProgramResources program, + SharedLayoutTestBinding descriptors, (float x, float y) jitterPx, Func sceneAt) + { + var inputs = CreateInputSet(textures); + UploadRgba16F(textures, inputs.SceneTex, + (x, _) => sceneAt(x), (x, _) => sceneAt(x), (x, _) => sceneAt(x), (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + UploadFlatRgba16F(textures, inputs.HistoryColor, 0f, 0f, 0f, 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + var uniforms = new TaaUniforms { ResetHistory = 1, JitterPx = { [0] = jitterPx.x, [1] = jitterPx.y } }; + + ResolveOnce(context, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context, commands, textures, output.Color, 8); + var columnAverage = new float[Size]; + for (int x = 0; x < Size; x++) + { + float sum = 0f; + for (int y = 0; y < Size; y++) sum += ReadHalf(colorBytes, x, y, 0, 8); + columnAverage[x] = sum / Size; + } + return FindThresholdCrossing(columnAverage, 0.5f); + } + + /// + /// With LINEAR history sampling and a uniform +0.5px motion, a + /// one-texel-wide bright column in historyGlow lands, in texel + /// space, exactly on the boundary between two texels for two adjacent + /// output columns: raster column x0 samples 50% texel x0 / 50% texel + /// x0+1, and column x0-1 samples 50% texel x0-1 / 50% texel x0. Both + /// should read half the bright value if - and only if - the device is + /// actually doing bilinear filtering on that sampler, not point + /// sampling. glowTex's resolve path (mix(historyGlowSample, + /// glow, alpha)) has no neighbourhood clip of its own, unlike colour, + /// so it isolates the sampler behaviour cleanly. + /// + [SkippableFact] + public unsafe void LinearHistorySamplingSpreadsAOnePixelLineOverTwoColumns() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const int brightColumn = 16; + var inputs = CreateInputSet(textures); + UploadFlatRgba16F(textures, inputs.SceneTex, 0f, 0f, 0f, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + // Written +0.5px motion (matches depth), so historyUv reads + // (pixelCentre + 0.5) * invSize - a half-texel shift. + UploadFlatRgba16F(textures, inputs.MotionTex, 0.5f, 0f, 0f, 0.5f); + UploadFlatRgba16F(textures, inputs.HistoryColor, 0f, 0f, 0f, 1f); + UploadRgba8(textures, inputs.HistoryGlow, + (x, _) => x == brightColumn ? (byte)255 : (byte)0, + (x, _) => x == brightColumn ? (byte)255 : (byte)0, + (x, _) => x == brightColumn ? (byte)255 : (byte)0, + (_, _) => (byte)255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + // Heavy history weight so resolvedGlow ~= historyGlowSample. + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.02f }; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] glowBytes = ReadTextureBytes(context!, commands, textures, output.Glow, 4); + float below = ReadByteChannel(glowBytes, brightColumn - 1, (int)Size / 2, 0); + float at = ReadByteChannel(glowBytes, brightColumn, (int)Size / 2, 0); + float farBackground = ReadByteChannel(glowBytes, brightColumn - 8, (int)Size / 2, 0); + + _output.WriteLine($"column {brightColumn - 1}={below}, column {brightColumn}={at}, background={farBackground}"); + + // Both straddling columns should read roughly half the bright + // value - not one at full brightness and the other at zero, + // which is what point/nearest sampling would produce. + Assert.InRange(below, 0.30f, 0.70f); + Assert.InRange(at, 0.30f, 0.70f); + Assert.True(Math.Abs(below - at) < 0.15f, + $"the two straddling columns should read close to equal (bilinear midpoint), got {below} vs {at}"); + Assert.True(farBackground < 0.1f, "a column away from the line should stay near background"); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// + /// A NaN anywhere in the history colour sample must be treated exactly + /// like resetHistory=1: the shader's own comment says NaN "survives any + /// weighted blend, poisoning the pixel forever", so it is detected and + /// swapped for the current frame's values with full current weight. Here + /// resetHistory stays 0 and blendAlpha is a normal 0.1 - only the NaN + /// planted in the history colour texture should force the reset. + /// + [SkippableFact] + public unsafe void NanInHistoryIsTreatedAsAReset() + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + const float currentR = 0.65f, currentG = 0.4f, currentB = 0.25f; + var inputs = CreateInputSet(textures); + UploadFlatRgba16F(textures, inputs.SceneTex, currentR, currentG, currentB, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + // NaN in history colour - nothing else in the history is broken. + UploadFlatRgba16F(textures, inputs.HistoryColor, float.NaN, float.NaN, float.NaN, 1f); + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.HistoryDepth, 0.5f); + + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.1f }; + + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + float r = ReadHalf(colorBytes, x, y, 0, 8); + float g = ReadHalf(colorBytes, x, y, 1, 8); + float b = ReadHalf(colorBytes, x, y, 2, 8); + Assert.False(float.IsNaN(r) || float.IsNaN(g) || float.IsNaN(b), + $"NaN leaked into the output at ({x},{y})"); + Assert.InRange(r, currentR - 0.02f, currentR + 0.02f); + Assert.InRange(g, currentG - 0.02f, currentG + 0.02f); + Assert.InRange(b, currentB - 0.02f, currentB + 0.02f); + } + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + // ------------------------------- 2026-09-11: distant foliage jitter was the resolve + // + // The four tests below pin the fix for the distant-foliage flicker (docs/vulkan.md + // "Follow-up 2026-09-11"): a single-sample disocclusion test rejected history on + // ~3.7% of distant leaf pixels per frame and a fixed blend weight let the clip box + // drag the history; the 3x3 nearest-depth test and the anti-flicker weight took it + // to ~1.1%. Do not revert either half. The temporal ones run several resolves + // ping-ponged between two history sets with no readback inside the loop (this + // harness has no swapchain; each resolve is its own submitted frame). + + /// + /// Anti-flicker current weight: a pixel that survived rejection takes + /// mix(1.2, 0.3, w * w) * blendAlpha of the current frame, with + /// w = 1 - |lumCur - lumHist| / max(lumCur, max(lumHist, 0.2)) on the rectified + /// luminance. Columns cycle 0.5 / 1.0 / 0.0, so every 3x3 box spans [0, 1] and no + /// history value in play is clipped. Class 0.5 starts converged (history equals + /// current: w = 1, weight 0.3 x blendAlpha); class 1.0 starts from a history of 0 + /// (w = 0, weight 1.2 x blendAlpha). Glow blends as mix(historyGlow, glow, alpha) + /// with no clip and no luminance weighting, so a glow step from 0 to 1 reads the + /// weight itself; the colour step response is checked against the same + /// recurrence. One frame and four frames, each also required to sit clearly apart + /// from what the old fixed blendAlpha weight produces. + /// + [SkippableFact] + public void AntiFlickerWeightsFollowTheLuminanceDifference() + { + const float blendAlpha = 0.1f; + // The model rounds the UNORM8 glow per frame exactly as the target does, so it + // matches to the LSB; 1.5 LSB leaves room for rounding at .5 only. The fixed + // weight is 5 LSB away in the closest case (large change, one frame). + const double glowTolerance = 1.5 / 255.0, colourTolerance = 0.004; + static float Scene(int x) => (x % 3) switch { 0 => 0.5f, 1 => 1.0f, _ => 0.0f }; + static float Seed(int x) => x % 3 == 0 ? 0.5f : 0.0f; + + foreach (int frames in new[] { 1, 4 }) + { + TemporalRun? run = RunTemporal(frames, new TaaUniforms { BlendAlpha = blendAlpha }, + (textures, history) => + { + UploadRgba16F(textures, history.Color, (x, _) => Seed(x), (x, _) => Seed(x), (x, _) => Seed(x), (_, _) => 1f); + UploadFlatRgba8(textures, history.Glow, 0, 0, 0, 255); + // Identity camera: window depth 0.5 is linear depth 0, what the resolve writes back. + UploadFlatR32F(textures, history.Depth, 0f); + }, + (frame, textures, inputs) => + { + if (frame > 0) return; + UploadRgba16F(textures, inputs.SceneTex, (x, _) => Scene(x), (x, _) => Scene(x), (x, _) => Scene(x), (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 255, 255, 255, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + }); + Skip.If(run == null, "No usable Vulkan device."); + + foreach ((string name, int column, float current, float seed) in new[] + { + ("converged", 0, 0.5f, 0.5f), + ("large change", 1, 1.0f, 0.0f), + }) + { + (double colour, double glow) expected = AntiFlickerModel(current, seed, frames, blendAlpha, antiFlicker: true); + (double colour, double glow) fixedWeight = AntiFlickerModel(current, seed, frames, blendAlpha, antiFlicker: false); + double worstGlow = 0, worstColour = 0, glowAtCentre = 0; + for (int y = 2; y < Size - 2; y++) + for (int x = 3; x < Size - 3; x++) + { + if (x % 3 != column) continue; + double glow = ReadByteChannel(run!.Glow, x, y, 0); + if (y == Size / 2) glowAtCentre = glow; + worstGlow = Math.Max(worstGlow, Math.Abs(glow - expected.glow)); + worstColour = Math.Max(worstColour, Math.Abs(ReadHalf(run.Color, x, y, 0, 8) - expected.colour)); + } + _output.WriteLine($"{frames} frame(s), {name}: glow {glowAtCentre:F4} (anti-flicker model {expected.glow:F4}, fixed-weight model {fixedWeight.glow:F4}), " + + $"colour model {expected.colour:F4}; worst glow error {worstGlow:F4}, worst colour error {worstColour:F4}"); + + Assert.True(Math.Abs(expected.glow - fixedWeight.glow) > 2 * glowTolerance, + $"{name}: the case cannot tell the anti-flicker weight from a fixed one"); + Assert.True(worstGlow <= glowTolerance, + $"{name}, {frames} frame(s): glow is {worstGlow:F4} off the anti-flicker step response"); + Assert.True(worstColour <= colourTolerance, + $"{name}, {frames} frame(s): colour is {worstColour:F4} off the anti-flicker step response"); + + if (frames == 1) + { + double weight = column == 0 ? 0.3 * blendAlpha : 1.2 * blendAlpha; + Assert.InRange(glowAtCentre, weight - glowTolerance, weight + glowTolerance); + } + } + } + } + + /// + /// A sub-pixel leaf in front of a far background lands in pixel P in one jitter + /// phase and in P + (1, 0) in the next, so the depth of both pixels flips between + /// 8 and 120 blocks every frame. The old single-sample test reset both pixels + /// every frame; the 3x3 nearest-depth test sees the leaf in both windows and keeps + /// the history. Glow marks the last three frames (R = last, G = the one before, + /// B = the one before that, each 1 only in its frame): a reset in frame k leaves + /// that channel at 1 - (weights after k), a kept pixel at weight * (1 - ...). + /// The same scenario runs on the shipped shader and on a copy with only the + /// disocclusion line put back to the old per-sample comparison. + /// + [SkippableFact] + public void FlippingSubPixelLeafKeepsItsHistory() + { + TemporalRun? nearest = RunLeafFlip(null); + Skip.If(nearest == null, "No usable Vulkan device."); + TemporalRun perSample = RunLeafFlip(WithPerSampleDisocclusion)!; + + foreach (int x in new[] { LeafX, LeafX + 1 }) + { + float r = ReadByteChannel(nearest!.Glow, x, LeafY, 0); + float g = ReadByteChannel(nearest.Glow, x, LeafY, 1); + float b = ReadByteChannel(nearest.Glow, x, LeafY, 2); + float pr = ReadByteChannel(perSample.Glow, x, LeafY, 0); + float pg = ReadByteChannel(perSample.Glow, x, LeafY, 1); + float pb = ReadByteChannel(perSample.Glow, x, LeafY, 2); + _output.WriteLine($"pixel ({x},{LeafY}): 3x3 nearest glow=({r:F3},{g:F3},{b:F3}), per-sample glow=({pr:F3},{pg:F3},{pb:F3})"); + + // The old test: reset in the last frame (every channel equals that frame's glow). + Assert.True(pr >= 0.99f && pg <= 0.01f && pb <= 0.01f, + $"the per-sample reference no longer resets the flipping pixel ({x},{LeafY}); the scenario does not reproduce the bug"); + // The shipped test: kept in each of the last three frames. + Assert.InRange(r, 0.015f, 0.25f); + Assert.InRange(g, 0.015f, 0.25f); + Assert.InRange(b, 0.015f, 0.25f); + } + + // Not asserted, printed for the record: the background pixels just outside the + // leaf's two positions see it enter their current 3x3 while their history + // window never held it, so the nearest-depth test resets them instead. + foreach (int x in new[] { LeafX - 1, LeafX + 2 }) + { + _output.WriteLine($"fringe pixel ({x},{LeafY}): 3x3 nearest R={ReadByteChannel(nearest!.Glow, x, LeafY, 0):F3}, per-sample R={ReadByteChannel(perSample.Glow, x, LeafY, 0):F3}"); + } + } + + /// + /// The 3x3 test must not hide a real disocclusion: an 8x8 block at 8 blocks in + /// front of a background at 120 is present for three frames and gone in the + /// fourth. Every pixel of the block's interior has only background in its current + /// 3x3 and only the block in its history 3x3, so it resets (glow marks the last + /// frame: 1 after a reset) and shows the background colour at once. A pixel far + /// from the block keeps its history with the converged weight 0.3 x blendAlpha. + /// + [SkippableFact] + public void DisocclusionLargerThanTheNeighbourhoodStillResets() + { + const int frames = 4; + const float blendAlpha = 0.1f, blockColour = 0.9f, backgroundColour = 0.4f; + PerspectiveCamera camera = CreatePerspective(); + float nearDepth = camera.WindowDepth(LeafLinearDepth), farDepth = camera.WindowDepth(BackgroundLinearDepth); + static bool InBlock(int x, int y) => x >= 12 && x < 20 && y >= 12 && y < 20; + + TemporalRun? run = RunTemporal(frames, camera.Uniforms(blendAlpha), + (textures, history) => + { + UploadRgba16F(textures, history.Color, (x, y) => InBlock(x, y) ? blockColour : backgroundColour, + (x, y) => InBlock(x, y) ? blockColour : backgroundColour, + (x, y) => InBlock(x, y) ? blockColour : backgroundColour, (_, _) => 1f); + UploadFlatRgba8(textures, history.Glow, 0, 0, 0, 255); + UploadR32F(textures, history.Depth, (x, y) => InBlock(x, y) ? LeafLinearDepth : BackgroundLinearDepth); + }, + (frame, textures, inputs) => + { + bool present = frame < frames - 1; + bool Block(int x, int y) => present && InBlock(x, y); + UploadRgba16F(textures, inputs.SceneTex, (x, y) => Block(x, y) ? blockColour : backgroundColour, + (x, y) => Block(x, y) ? blockColour : backgroundColour, + (x, y) => Block(x, y) ? blockColour : backgroundColour, (_, _) => 1f); + UploadR32F(textures, inputs.DepthTex, (x, y) => Block(x, y) ? nearDepth : farDepth); + UploadRgba16F(textures, inputs.MotionTex, (_, _) => 0f, (_, _) => 0f, (_, _) => 0f, + (x, y) => Block(x, y) ? nearDepth : farDepth); + byte mark = frame == frames - 1 ? (byte)255 : (byte)0; + UploadFlatRgba8(textures, inputs.GlowTex, mark, mark, mark, 255); + }); + Skip.If(run == null, "No usable Vulkan device."); + + for (int y = 13; y < 19; y++) + for (int x = 13; x < 19; x++) + { + float glow = ReadByteChannel(run!.Glow, x, y, 0); + float colour = ReadHalf(run.Color, x, y, 0, 8); + Assert.True(glow >= 254f / 255f, $"disoccluded pixel ({x},{y}) kept its history: glow {glow:F3}"); + Assert.InRange(colour, backgroundColour - 0.01f, backgroundColour + 0.01f); + } + + foreach ((int x, int y) in new[] { (4, 4), (27, 27) }) + { + float glow = ReadByteChannel(run!.Glow, x, y, 0); + _output.WriteLine($"control pixel ({x},{y}): glow {glow:F4}, converged weight {0.3 * blendAlpha:F4}"); + Assert.InRange(glow, 0.3f * blendAlpha - 2.5f / 255f, 0.3f * blendAlpha + 2.5f / 255f); + } + } + + /// + /// At a depth edge the motion vector comes from the nearest-depth tap of the 3x3. + /// Foreground (8 blocks, columns >= 16) moved left by 4 px, so it writes + /// mv = (+4, 0); the background (120 blocks) is static. Column 15 is background, + /// but its 3x3 holds foreground taps, so it reprojects with +4 and reads history + /// column 19, the only bright column of the history glow. Column 14 (no foreground + /// in its 3x3) reads its own column, column 16 (foreground) reads column 20: both + /// dark. A resolve that used the pixel's own vector would leave column 15 dark. + /// + [SkippableFact] + public void MotionComesFromTheNearestDepthTapAtAnEdge() + { + const int edge = 16, shift = 4; + PerspectiveCamera camera = CreatePerspective(); + float nearDepth = camera.WindowDepth(LeafLinearDepth), farDepth = camera.WindowDepth(BackgroundLinearDepth); + + TemporalRun? run = RunTemporal(1, camera.Uniforms(0.05f), + (textures, history) => + { + UploadFlatRgba16F(textures, history.Color, 0.5f, 0.5f, 0.5f, 1f); + UploadRgba8(textures, history.Glow, (x, _) => x == edge - 1 + shift ? (byte)255 : (byte)0, + (_, _) => 0, (_, _) => 0, (_, _) => 255); + // Last frame the foreground started at column edge + shift. + UploadR32F(textures, history.Depth, (x, _) => x >= edge + shift ? LeafLinearDepth : BackgroundLinearDepth); + }, + (_, textures, inputs) => + { + UploadFlatRgba16F(textures, inputs.SceneTex, 0.5f, 0.5f, 0.5f, 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadR32F(textures, inputs.DepthTex, (x, _) => x >= edge ? nearDepth : farDepth); + UploadRgba16F(textures, inputs.MotionTex, (x, _) => x >= edge ? shift : 0f, (_, _) => 0f, (_, _) => 0f, + (x, _) => x >= edge ? nearDepth : farDepth); + }); + Skip.If(run == null, "No usable Vulkan device."); + + for (int y = 4; y < Size - 4; y++) + { + float edgePixel = ReadByteChannel(run!.Glow, edge - 1, y, 0); + float background = ReadByteChannel(run.Glow, edge - 2, y, 0); + float foreground = ReadByteChannel(run.Glow, edge, y, 0); + if (y == Size / 2) + _output.WriteLine($"row {y}: column {edge - 2} glow {background:F3}, column {edge - 1} glow {edgePixel:F3}, column {edge} glow {foreground:F3}"); + Assert.True(edgePixel >= 0.9f, $"column {edge - 1} row {y} did not reproject with the nearest tap's vector (glow {edgePixel:F3})"); + Assert.True(background <= 0.05f, $"column {edge - 2} row {y} moved although its 3x3 holds no foreground (glow {background:F3})"); + Assert.True(foreground <= 0.05f, $"column {edge} row {y} did not use its own vector (glow {foreground:F3})"); + } + } + + private const int LeafX = 16, LeafY = 16; + private const float LeafLinearDepth = 8f, BackgroundLinearDepth = 120f; + + /// The disocclusion line the shipped resolve carries, and the per-sample line it replaced. + private const string NearestDepthRejection = + "bool depthMismatch = abs(historyNearest - closestLinearDepth) > depthTolerance;"; + private const string PerSampleRejection = + "bool depthMismatch = abs(historyLinear - linearDepth) > 0.5 + 0.08 * linearDepth;"; + + private static string WithPerSampleDisocclusion(string fragment) + { + Assert.Contains(NearestDepthRejection, fragment); + return fragment.Replace(NearestDepthRejection, PerSampleRejection, StringComparison.Ordinal); + } + + /// A sparse distant leaf keeps clipped history, while a solid + /// silhouette and a flat-depth disocclusion reset it. + [SkippableFact] + public void SparseDistantEdgeKeepsHistoryWhileSolidDisocclusionResets() + { + PerspectiveCamera camera = CreatePerspective(); + float nearDepth = camera.WindowDepth(80f); + float farDepth = camera.WindowDepth(120f); + + TemporalRun? Run(bool edge, bool solid = false) => RunTemporal(1, camera.Uniforms(0.1f), + (textures, history) => + { + UploadFlatRgba16F(textures, history.Color, 0.5f, 0.5f, 0.5f, 1f); + UploadFlatRgba8(textures, history.Glow, 255, 0, 0, 255); + UploadFlatR32F(textures, history.Depth, 120f); + }, + (_, textures, inputs) => + { + Func colour = (x, y) => x == LeafX && y == LeafY ? 0.9f : 0.5f; + UploadRgba16F(textures, inputs.SceneTex, colour, colour, colour, (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadR32F(textures, inputs.DepthTex, + (x, y) => !edge || (x == LeafX && y == LeafY) || (solid && x <= LeafX) + ? nearDepth : farDepth); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, nearDepth); + }); + + TemporalRun? silhouette = Run(edge: true); + TemporalRun? solid = Run(edge: true, solid: true); + TemporalRun? flat = Run(edge: false); + Skip.If(silhouette == null || solid == null || flat == null, "No usable Vulkan device."); + float retained = ReadHalf(silhouette!.Color, LeafX, LeafY, 0, 8); + float solidReset = ReadHalf(solid!.Color, LeafX, LeafY, 0, 8); + float reset = ReadHalf(flat!.Color, LeafX, LeafY, 0, 8); + Assert.True(retained < 0.75f, $"distant edge discarded clipped colour history ({retained:F3})"); + Assert.True(solidReset > 0.85f, $"solid distant edge kept stale colour history ({solidReset:F3})"); + Assert.True(reset > 0.85f, $"flat-depth disocclusion kept colour history ({reset:F3})"); + Assert.True(ReadByteChannel(silhouette.Glow, LeafX, LeafY, 0) < 0.05f, + "distant edge carried stale glow into a new surface"); + Assert.True(ReadByteChannel(flat.Glow, LeafX, LeafY, 0) < 0.05f, + "flat-depth disocclusion carried stale glow"); + } + + private TemporalRun? RunLeafFlip(Func? fragmentTransform) + { + const int frames = 6; + const float leafColour = 0.9f, backgroundColour = 0.4f; + PerspectiveCamera camera = CreatePerspective(); + float nearDepth = camera.WindowDepth(LeafLinearDepth), farDepth = camera.WindowDepth(BackgroundLinearDepth); + + return RunTemporal(frames, camera.Uniforms(0.1f), + (textures, history) => + { + // The phase before frame 0: the leaf sat in the right-hand pixel. + bool Leaf(int x, int y) => x == LeafX + 1 && y == LeafY; + UploadRgba16F(textures, history.Color, (x, y) => Leaf(x, y) ? leafColour : backgroundColour, + (x, y) => Leaf(x, y) ? leafColour : backgroundColour, + (x, y) => Leaf(x, y) ? leafColour : backgroundColour, (_, _) => 1f); + UploadFlatRgba8(textures, history.Glow, 0, 0, 0, 255); + UploadR32F(textures, history.Depth, (x, y) => Leaf(x, y) ? LeafLinearDepth : BackgroundLinearDepth); + }, + (frame, textures, inputs) => + { + int leafX = frame % 2 == 0 ? LeafX : LeafX + 1; + bool Leaf(int x, int y) => x == leafX && y == LeafY; + UploadRgba16F(textures, inputs.SceneTex, (x, y) => Leaf(x, y) ? leafColour : backgroundColour, + (x, y) => Leaf(x, y) ? leafColour : backgroundColour, + (x, y) => Leaf(x, y) ? leafColour : backgroundColour, (_, _) => 1f); + UploadR32F(textures, inputs.DepthTex, (x, y) => Leaf(x, y) ? nearDepth : farDepth); + UploadRgba16F(textures, inputs.MotionTex, (_, _) => 0f, (_, _) => 0f, (_, _) => 0f, + (x, y) => Leaf(x, y) ? nearDepth : farDepth); + UploadFlatRgba8(textures, inputs.GlowTex, + frame == frames - 1 ? (byte)255 : (byte)0, + frame == frames - 2 ? (byte)255 : (byte)0, + frame == frames - 3 ? (byte)255 : (byte)0, 255); + }, + fragmentTransform); + } + + /// + /// The resolve's colour and glow recurrence for a grey pixel whose history stays + /// inside the neighbourhood box, with the glow stored as UNORM8 every frame. + /// + private static (double colour, double glow) AntiFlickerModel( + double current, double history, int frames, double blendAlpha, bool antiFlicker) + { + double glow = 0; + for (int i = 0; i < frames; i++) + { + double alpha = blendAlpha; + if (antiFlicker) + { + double w = 1.0 - Math.Abs(current - history) / Math.Max(current, Math.Max(history, 0.2)); + alpha = blendAlpha * 1.2 + (blendAlpha * 0.3 - blendAlpha * 1.2) * w * w; + } + double wCur = alpha / (1.0 + current), wHist = (1.0 - alpha) / (1.0 + history); + history = (current * wCur + history * wHist) / Math.Max(wCur + wHist, 1e-5); + glow = Math.Round((glow * (1.0 - alpha) + alpha) * 255.0) / 255.0; + } + return (history, glow); + } + + /// + /// A converged surface under per-frame stochastic noise (the GTAO term's residual after its + /// spatial denoise: XeGTAO relies on TAA for the temporal part). A static camera, zero motion, + /// a mid-grey surface and +-noiseAmplitude white noise that changes every frame. After 24 + /// resolves the history must carry far less of that noise than one frame does: with alpha + /// 0.1 and the anti-flicker weighting the expected residual is about a seventh of the input. + /// + /// The textured rows report, without asserting, how much a static +-textureAmplitude per-texel + /// pattern is distorted by the neighbourhood clip (a texel further than varianceGamma sigma + /// from its 3x3 mean is clamped toward that mean every frame): measured 2026-09-17 at ~3 % for + /// a +-5 % texture, independent of the temporal noise. That is the known cost of variance + /// clipping on fine detail, not a convergence failure. + /// + [SkippableTheory] + [InlineData(0.03f, 0f)] + [InlineData(0.10f, 0f)] + [InlineData(0.03f, 0.05f)] + [InlineData(0.10f, 0.05f)] + public unsafe void PerFrameNoiseOnAStaticSurfaceAveragesOut(float noiseAmplitude, float textureAmplitude) + { + const float baseValue = 0.35f; + var random = new Random(12345); + float[] texture = new float[Size * Size]; + for (int i = 0; i < texture.Length; i++) texture[i] = baseValue + (float)((random.NextDouble() - 0.5) * 2.0 * textureAmplitude); + var frames = new List(); + for (int f = 0; f < 24; f++) + { + float[] frame = new float[Size * Size]; + for (int i = 0; i < frame.Length; i++) + frame[i] = texture[i] * (1f + noiseAmplitude * (float)((random.NextDouble() - 0.5) * 2.0)); + frames.Add(frame); + } + + var uniforms = new TaaUniforms { ResetHistory = 0, BlendAlpha = 0.1f }; + TemporalRun? run = RunTemporal(24, uniforms, + (textures, history) => + { + UploadFlatRgba16F(textures, history.Color, baseValue, baseValue, baseValue, 1f); + UploadFlatRgba8(textures, history.Glow, 0, 0, 0, 255); + UploadFlatR32F(textures, history.Depth, 0.5f); + }, + (frame, textures, inputs) => + { + float[] values = frames[frame]; + Func value = (x, y) => values[y * Size + x]; + UploadRgba16F(textures, inputs.SceneTex, value, value, value, (x, y) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 0.5f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0.5f); + }); + Skip.If(run == null, "No usable Vulkan device."); + + // Residual: history minus the noise-free texture, over the interior. + double sum = 0, sum2 = 0; int count = 0; + double inSum2 = 0; + for (int y = 2; y < Size - 2; y++) + for (int x = 2; x < Size - 2; x++) + { + float t = texture[y * Size + x]; + float h = ReadHalf(run!.Color, x, y, 0, 8); + double e = (h - t) / t; + sum += e; sum2 += e * e; count++; + double n = (frames[23][y * Size + x] - t) / t; + inSum2 += n * n; + } + double residual = Math.Sqrt(sum2 / count - (sum / count) * (sum / count)); + double input = Math.Sqrt(inSum2 / count); + _output.WriteLine($"noise {noiseAmplitude} texture {textureAmplitude}: one-frame relative noise {input:F4}, history residual {residual:F4}, ratio {residual / input:F3}, bias {sum / count:F4}"); + if (textureAmplitude == 0f) + { + Assert.True(residual / input < 0.34, $"the resolve kept {residual / input:F2} of the per-frame noise"); + } + } + + private sealed class TemporalRun + { + public byte[] Color = Array.Empty(); + public byte[] Glow = Array.Empty(); + } + + /// + /// A fresh context, resolves ping-ponged between two + /// history sets (the first seeded by , inputs + /// uploaded per frame by ), and one readback of the + /// last write after the loop. Null when there is no usable device. + /// + private TemporalRun? RunTemporal(int frames, TaaUniforms uniforms, + Action seedHistory, + Action uploadFrame, + Func? fragmentTransform = null) + { + var messages = new List(); + if (!TryCreateContext(_output, messages, out VulkanContext? context)) return null; + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state, fragmentTransform); + + var inputs = CreateInputSet(textures); + TaaAttachmentSet history = CreateAttachmentSet(textures, targets); + TaaAttachmentSet current = CreateAttachmentSet(textures, targets); + seedHistory(textures, history); + + for (int frame = 0; frame < frames; frame++) + { + uploadFrame(frame, textures, inputs); + inputs.HistoryColor = history.Color; + inputs.HistoryGlow = history.Glow; + inputs.HistoryDepth = history.Depth; + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, current); + (history, current) = (current, history); + } + + var run = new TemporalRun + { + Color = ReadTextureBytes(context!, commands, textures, history.Color, 8), + Glow = ReadTextureBytes(context!, commands, textures, history.Glow, 4), + }; + + ValidationAssert.NoErrors(messages); + ValidationAssert.NoSyncHazards(messages); + return run; + } + } + + /// + /// A real perspective (fov 70, near 0.1, far 200) with an identity view, so the + /// resolve's linear depth is the distance the test names and nearer window depth + /// means nearer linear depth, as in the game. Motion is always written in these + /// tests, so the matrices only feed the depth reconstruction. + /// + private sealed class PerspectiveCamera + { + private readonly double[] _projection; + public readonly float[] Projection; + public readonly float[] InverseProjection; + + public PerspectiveCamera(double[] projection, double[] inverse) + { + _projection = projection; + Projection = Array.ConvertAll(projection, v => (float)v); + InverseProjection = Array.ConvertAll(inverse, v => (float)v); + } + + public float WindowDepth(double linear) + { + double clipZ = _projection[10] * -linear + _projection[14]; + double clipW = _projection[11] * -linear + _projection[15]; + return (float)(clipZ / clipW * 0.5 + 0.5); + } + + public TaaUniforms Uniforms(float blendAlpha) => new() + { + BlendAlpha = blendAlpha, + InvViewProjJittered = InverseProjection, + PrevViewProj = Projection, + ViewMatrix = Identity4, + }; + } + + private static PerspectiveCamera CreatePerspective() + { + const double near = 0.1, far = 200.0, fov = 70.0 * Math.PI / 180.0; + double[] projection = Vintagestory.API.MathTools.Mat4d.Perspective(Vintagestory.API.MathTools.Mat4d.Create(), fov, 1.0, near, far); + double[] inverse = Vintagestory.API.MathTools.Mat4d.Invert(Vintagestory.API.MathTools.Mat4d.Create(), projection)!; + return new PerspectiveCamera(projection, inverse); + } + + // ------------------------------------------------------------------ setup + + private static ShaderProgramResources LoadProgram( + VulkanContext context, ShaderCompiler compiler, PipelineKeyState state, + Func? fragmentTransform = null) + { + Dictionary files = ShaderCorpus.LoadShaderFiles(); + if (fragmentTransform != null) + { + files["taa-resolve.fsh"] = fragmentTransform(files["taa-resolve.fsh"]); + } + Dictionary includes = ShaderCorpus.LoadIncludes(); + List stages = ShaderCorpus.BuildProgram( + "taa-resolve", files, includes, ShaderCorpus.Variants().First()); + Assert.NotEmpty(stages); + + TranslatedProgram translated = ShaderTranslator.Translate(stages, compiler); + Assert.True(translated.Success, string.Join("; ", translated.Errors)); + + var program = new ShaderProgramResources(context, 1, translated); + state.SetProgram(1); + return program; + } + + /// + /// The resolve's own sky path had the same trap the sky pass had: it treated + /// the reconstructed far point (in the origin space CameraMatrixOrigin draws + /// in, where the eye sits at LocalEyePos) as the view direction. A still + /// camera above the origin then reprojected every sky pixel by a fixed + /// eye / far * (rows / 2) / tan(fov / 2) pixels - 0.6 px in game - and the + /// history drifted by that much every frame. With a real perspective and a + /// translated view the band has to stay exactly where it is, horizontally + /// and vertically (the in-game error was vertical: the eye offset is on Y). + /// + [SkippableTheory] + [InlineData(1.7)] + [InlineData(6.0)] + public unsafe void SkyStaysPutWhenTheCameraSitsAboveTheOrigin(double eyeHeight) + { + var messages = new List(); + Skip.IfNot(TryCreateContext(_output, messages, out VulkanContext? context), "No usable Vulkan device."); + + const double near = 0.0689, far = 60.0, fov = 70.0 * Math.PI / 180.0; + double[] projection = Vintagestory.API.MathTools.Mat4d.Perspective(Vintagestory.API.MathTools.Mat4d.Create(), fov, 1.0, near, far); + double[] view = Vintagestory.API.MathTools.Mat4d.Identity(Vintagestory.API.MathTools.Mat4d.Create()); + view = Vintagestory.API.MathTools.Mat4d.RotateX(view, view, -0.2); + view = Vintagestory.API.MathTools.Mat4d.Translate(view, view, 0.0, -eyeHeight, 0.0); + double[] viewProj = Vintagestory.API.MathTools.Mat4d.Mul(Vintagestory.API.MathTools.Mat4d.Create(), projection, view); + double[] inverse = Vintagestory.API.MathTools.Mat4d.Invert(Vintagestory.API.MathTools.Mat4d.Create(), viewProj)!; + float[] invF = Array.ConvertAll(inverse, v => (float)v); + float[] vpF = Array.ConvertAll(viewProj, v => (float)v); + float[] viewF = Array.ConvertAll(view, v => (float)v); + double predictedBias = eyeHeight / far * (Size / 2.0) / Math.Tan(fov / 2.0); + _output.WriteLine($"eye {eyeHeight}: far-point-as-direction would drift the sky by ~{predictedBias:F2} px per frame"); + + using (context) + using (var commands = new SetupQueue(context!)) + using (var textures = new TextureManager(context!, commands.Uploads)) + { + var state = new PipelineKeyState(); + using var targets = new RenderTargetManager(context!, textures); + using var pipelines = new GraphicsPipelineCache(context!); + using var compiler = new ShaderCompiler(); + using var descriptors = new SharedLayoutTestBinding(context!, textures); + using ShaderProgramResources program = LoadProgram(context!, compiler, state); + + var inputs = CreateInputSet(textures); + // A checkered scene keeps the neighbourhood clip box wide (0.3..0.7), + // so the history stripe survives the rectification. + UploadRgba16F(textures, inputs.SceneTex, (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, + (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, (x, y) => ((x + y) % 2 == 0) ? 0.3f : 0.7f, (_, _) => 1f); + UploadFlatRgba8(textures, inputs.GlowTex, 0, 0, 0, 255); + UploadFlatR32F(textures, inputs.DepthTex, 1.0f); + UploadFlatRgba16F(textures, inputs.MotionTex, 0f, 0f, 0f, 0f); + + const int stripeStart = 14, stripeWidth = 4; + const float background = 0.5f, stripe = 1.0f; + UploadFlatRgba8(textures, inputs.HistoryGlow, 0, 0, 0, 255); + // Linear view depth of the far plane, so the disocclusion test passes. + UploadFlatR32F(textures, inputs.HistoryDepth, (float)far); + TaaAttachmentSet output = CreateAttachmentSet(textures, targets); + var uniforms = new TaaUniforms + { + ResetHistory = 0, + BlendAlpha = 0.05f, + InvViewProjJittered = invF, + PrevViewProj = vpF, + ViewMatrix = viewF, + CameraDelta = new[] { 0f, 0f, 0f }, + }; + + UploadRgba16F(textures, inputs.HistoryColor, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (x, _) => x is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, _) => 1f); + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + byte[] colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + double centroid = RedCentroidX(colorBytes, stripeStart - 8, stripeStart + stripeWidth + 8, background); + _output.WriteLine($"stripe centroid x = {centroid:F3} (expected {stripeStart + stripeWidth / 2.0:F1})"); + Assert.InRange(centroid, stripeStart + stripeWidth / 2.0 - 0.35, stripeStart + stripeWidth / 2.0 + 0.35); + + UploadRgba16F(textures, inputs.HistoryColor, + (_, y) => y is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, y) => y is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, y) => y is >= stripeStart and < stripeStart + stripeWidth ? stripe : background, + (_, _) => 1f); + ResolveOnce(context!, commands, textures, state, targets, pipelines, program, descriptors, + inputs, uniforms, output); + colorBytes = ReadTextureBytes(context!, commands, textures, output.Color, 8); + double centroidY = RedCentroidY(colorBytes, stripeStart - 8, stripeStart + stripeWidth + 8, background); + _output.WriteLine($"stripe centroid y = {centroidY:F3} (expected {stripeStart + stripeWidth / 2.0:F1})"); + Assert.InRange(centroidY, stripeStart + stripeWidth / 2.0 - 0.35, stripeStart + stripeWidth / 2.0 + 0.35); + Assert.True(predictedBias > 0.6, "the case is too weak to catch the eye-offset bug"); + + ValidationAssert.NoErrors(messages); + + ValidationAssert.NoSyncHazards(messages); + } + } + + /// Red-weighted column centroid above over [startX, endX), pixel-centre convention. + private static double RedCentroidX(byte[] colorBytes, int startX, int endX, float background) + { + double num = 0, den = 0; + for (int y = 4; y < Size - 4; y++) + for (int x = startX; x < endX; x++) + { + double w = Math.Max(0f, ReadHalf(colorBytes, x, y, 0, 8) - background); + num += w * (x + 0.5); den += w; + } + return den > 0 ? num / den : double.NaN; + } + + private static double RedCentroidY(byte[] colorBytes, int startY, int endY, float background) + { + double num = 0, den = 0; + for (int x = 4; x < Size - 4; x++) + for (int y = startY; y < endY; y++) + { + double w = Math.Max(0f, ReadHalf(colorBytes, x, y, 0, 8) - background); + num += w * (y + 0.5); den += w; + } + return den > 0 ? num / den : double.NaN; + } + + /// The seven sampler inputs the resolve declares. + private sealed class TaaInputSet + { + public int SceneTex; + public int GlowTex; + public int MotionTex; + public int DepthTex; + public int HistoryColor; + public int HistoryGlow; + public int HistoryDepth; + } + + /// One MRT write target: colour history, aux/glow, linear depth. + private sealed class TaaAttachmentSet + { + public int Color; + public int Glow; + public int Depth; + public int Framebuffer; + } + + private sealed class TaaUniforms + { + public float[] RenderSize = { Size, Size }; + public float[] JitterPx = { 0f, 0f }; + public float[] InvViewProjJittered = Identity4; + public float[] PrevViewProj = Identity4; + public float[] ViewMatrix = Identity4; + public float[] CameraDelta = { 0f, 0f, 0f }; + public int ResetHistory; + public float BlendAlpha = 0.1f; + public float VarianceGamma = 1.25f; + } + + private static TaaInputSet CreateInputSet(TextureManager textures) => new() + { + SceneTex = textures.Create(Size, Size, Format.R16G16B16A16Sfloat), + GlowTex = textures.Create(Size, Size, Format.R8G8B8A8Unorm), + MotionTex = textures.Create(Size, Size, Format.R16G16B16A16Sfloat), + DepthTex = textures.Create(Size, Size, Format.R32Sfloat), + HistoryColor = textures.Create(Size, Size, Format.R16G16B16A16Sfloat), + HistoryGlow = textures.Create(Size, Size, Format.R8G8B8A8Unorm), + HistoryDepth = textures.Create(Size, Size, Format.R32Sfloat), + }; + + private static TaaAttachmentSet CreateAttachmentSet(TextureManager textures, RenderTargetManager targets) + { + var set = new TaaAttachmentSet + { + Color = textures.Create(Size, Size, Format.R16G16B16A16Sfloat), + Glow = textures.Create(Size, Size, Format.R8G8B8A8Unorm), + Depth = textures.Create(Size, Size, Format.R32Sfloat), + }; + set.Framebuffer = targets.Create(Size, Size); + targets.Attach(set.Framebuffer, 0, set.Color); + targets.Attach(set.Framebuffer, 1, set.Glow); + targets.Attach(set.Framebuffer, 2, set.Depth); + return set; + } + + // ------------------------------------------------------------------- draw + + /// + /// One resolve pass: writes the shadow buffer, uploads it to a dedicated + /// dynamic-uniform-buffer, builds set 0 (uniforms) and set 1 (samplers) by + /// hand through a private , and draws the + /// fullscreen triangle - the same three-step shape VulkanDevice's own + /// draw path follows, scoped to a single named-uniform, named-sampler pass. + /// + private static unsafe void ResolveOnce( + VulkanContext context, SetupQueue commands, TextureManager textures, PipelineKeyState state, + RenderTargetManager targets, GraphicsPipelineCache pipelines, ShaderProgramResources program, + SharedLayoutTestBinding descriptors, TaaInputSet inputs, TaaUniforms uniforms, TaaAttachmentSet output) + { + SetUniformFloats(program, "renderSize", uniforms.RenderSize); + SetUniformFloats(program, "jitterPx", uniforms.JitterPx); + SetUniformFloats(program, "invViewProjJittered", uniforms.InvViewProjJittered); + SetUniformFloats(program, "prevViewProj", uniforms.PrevViewProj); + SetUniformFloats(program, "viewMatrix", uniforms.ViewMatrix); + SetUniformFloats(program, "cameraDelta", uniforms.CameraDelta); + SetUniformInt(program, "resetHistory", uniforms.ResetHistory); + SetUniformFloats(program, "blendAlpha", new[] { uniforms.BlendAlpha }); + SetUniformFloats(program, "varianceGamma", new[] { uniforms.VarianceGamma }); + + ulong shadowSize = (ulong)Math.Max(program.UniformShadow.Length, 16); + using var uniformBuffer = new VulkanBuffer(context, shadowSize, + BufferUsageFlags.UniformBufferBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + fixed (byte* source = program.UniformShadow) + { + System.Buffer.MemoryCopy(source, (void*)uniformBuffer.Mapped, + (long)shadowSize, program.UniformShadow.Length); + } + + var textureByName = new Dictionary(StringComparer.Ordinal) + { + ["sceneTex"] = inputs.SceneTex, + ["glowTex"] = inputs.GlowTex, + ["motionTex"] = inputs.MotionTex, + ["depthTex"] = inputs.DepthTex, + ["historyColor"] = inputs.HistoryColor, + ["historyGlow"] = inputs.HistoryGlow, + ["historyDepth"] = inputs.HistoryDepth, + }; + + var samplerState = SamplerState.Default with + { + MagFilter = Filter.Linear, + MinFilter = Filter.Linear, + AddressU = SamplerAddressMode.ClampToEdge, + AddressV = SamplerAddressMode.ClampToEdge, + }; + + var samplers = new Dictionary(StringComparer.Ordinal); + var sampledTextures = new VulkanTexture[program.Interface.Samplers.Count]; + for (int i = 0; i < sampledTextures.Length; i++) + { + SamplerBinding declared = program.Interface.Samplers[i]; + VulkanTexture texture = textures.Get(textureByName[declared.Name]) + ?? throw new InvalidOperationException("no texture bound for sampler '" + declared.Name + "'"); + sampledTextures[i] = texture; + samplers[declared.Name] = new SharedLayoutTestBinding.SampledTexture(textureByName[declared.Name], samplerState); + } + + VulkanFramebuffer bound = targets.Get(output.Framebuffer)!; + int formatsId = targets.FormatsIdOf(bound); + RenderTargetFormats formats = targets.FormatsOf(formatsId); + int attachmentCount = targets.EnabledAttachmentCount(bound); + + var blend = new AttachmentBlend[Math.Max(formats.ColorFormats.Length, 1)]; + for (int i = 0; i < blend.Length; i++) blend[i] = state.BlendFor(i); + + Pipeline pipeline = pipelines.Get( + state.BuildKey(0, formatsId, attachmentCount), + new GraphicsPipelineCache.PipelineRequest + { + Program = program, + VertexLayout = VertexLayoutDescription.Empty, + Targets = formats, + Blend = blend, + PolygonMode = state.PolygonMode, + Topology = state.Topology, + }); + + commands.SubmitAndWait(commandBuffer => + { + Vk api = context.Api; + + // Layout transitions cannot happen inside a rendering scope, so + // every sampled texture - including a previous iteration's output, + // still in ColorAttachmentOptimal - is put right before it opens. + foreach (VulkanTexture texture in sampledTextures) + { + textures.TransitionTexture(commandBuffer, texture, ImageLayout.ShaderReadOnlyOptimal); + } + descriptors.Transition(commandBuffer, Array.Empty()); + + targets.Bind(commandBuffer, output.Framebuffer); + targets.EnsureRendering(commandBuffer); + + api.CmdBindPipeline(commandBuffer, PipelineBindPoint.Graphics, pipeline); + + var viewport = new Viewport(0, 0, Size, Size, 0, 1); + api.CmdSetViewport(commandBuffer, 0, 1, &viewport); + var scissor = new Rect2D(new Offset2D(0, 0), new Extent2D(Size, Size)); + api.CmdSetScissor(commandBuffer, 0, 1, &scissor); + api.CmdSetCullMode(commandBuffer, CullModeFlags.None); + api.CmdSetFrontFace(commandBuffer, PipelineKeyState.FrontFace); + api.CmdSetPrimitiveTopology(commandBuffer, PrimitiveTopology.TriangleList); + api.CmdSetDepthTestEnable(commandBuffer, false); + api.CmdSetDepthWriteEnable(commandBuffer, false); + api.CmdSetDepthCompareOp(commandBuffer, CompareOp.Always); + api.CmdSetStencilTestEnable(commandBuffer, false); + api.CmdSetStencilOp(commandBuffer, StencilFaceFlags.FaceFrontAndBack, + StencilOp.Keep, StencilOp.Keep, StencilOp.Keep, CompareOp.Always); + api.CmdSetStencilCompareMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0xFF); + api.CmdSetStencilWriteMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0xFF); + api.CmdSetStencilReference(commandBuffer, StencilFaceFlags.FaceFrontAndBack, 0); + api.CmdSetLineWidth(commandBuffer, 1.0f); + + descriptors.Bind(commandBuffer, program, samplers, record: uniformBuffer); + + api.CmdDraw(commandBuffer, 3, 1, 0, 0); + targets.EndRendering(commandBuffer); + }); + } + + private static void SetUniformFloats(ShaderProgramResources program, string name, float[] values) + { + int location = program.LocationOf(name); + if (location < 0) return; + + var bytes = new byte[values.Length * sizeof(float)]; + for (int i = 0; i < values.Length; i++) + { + BitConverter.TryWriteBytes(bytes.AsSpan(i * sizeof(float), sizeof(float)), values[i]); + } + program.SetUniform(location, bytes); + } + + private static void SetUniformInt(ShaderProgramResources program, string name, int value) + { + int location = program.LocationOf(name); + if (location < 0) return; + program.SetUniform(location, BitConverter.GetBytes(value)); + } + + // --------------------------------------------------------------- textures + + private static unsafe void UploadFlatRgba16F( + TextureManager textures, int textureId, float r, float g, float b, float a) => + UploadRgba16F(textures, textureId, (_, _) => r, (_, _) => g, (_, _) => b, (_, _) => a); + + private static unsafe void UploadRgba16F( + TextureManager textures, int textureId, + Func r, Func g, Func b, Func a) + { + var data = new Half[Size * Size * 4]; + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + int i = (y * (int)Size + x) * 4; + data[i] = (Half)r(x, y); + data[i + 1] = (Half)g(x, y); + data[i + 2] = (Half)b(x, y); + data[i + 3] = (Half)a(x, y); + } + fixed (Half* pixels = data) + { + textures.Upload(textureId, 0, 0, 0, Size, Size, (IntPtr)pixels, 8); + } + } + + private static unsafe void UploadFlatRgba8( + TextureManager textures, int textureId, byte r, byte g, byte b, byte a) + { + var data = new byte[Size * Size * 4]; + for (int i = 0; i < data.Length; i += 4) + { + data[i] = r; data[i + 1] = g; data[i + 2] = b; data[i + 3] = a; + } + fixed (byte* pixels = data) + { + textures.Upload(textureId, 0, 0, 0, Size, Size, (IntPtr)pixels, 4); + } + } + + private static unsafe void UploadRgba8( + TextureManager textures, int textureId, + Func r, Func g, Func b, Func a) + { + var data = new byte[Size * Size * 4]; + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + int i = (y * (int)Size + x) * 4; + data[i] = r(x, y); + data[i + 1] = g(x, y); + data[i + 2] = b(x, y); + data[i + 3] = a(x, y); + } + fixed (byte* pixels = data) + { + textures.Upload(textureId, 0, 0, 0, Size, Size, (IntPtr)pixels, 4); + } + } + + private static unsafe void UploadFlatR32F(TextureManager textures, int textureId, float value) + { + var data = new float[Size * Size]; + Array.Fill(data, value); + fixed (float* pixels = data) + { + textures.Upload(textureId, 0, 0, 0, Size, Size, (IntPtr)pixels, 4); + } + } + + private static unsafe void UploadR32F(TextureManager textures, int textureId, Func value) + { + var data = new float[Size * Size]; + for (int y = 0; y < Size; y++) + for (int x = 0; x < Size; x++) + { + data[y * (int)Size + x] = value(x, y); + } + fixed (float* pixels = data) + { + textures.Upload(textureId, 0, 0, 0, Size, Size, (IntPtr)pixels, 4); + } + } + + // --------------------------------------------------------------- readback + + private static unsafe byte[] ReadTextureBytes( + VulkanContext context, SetupQueue commands, TextureManager textures, int textureId, int bytesPerPixel) + { + VulkanTexture texture = textures.Get(textureId)!; + ulong bytes = (ulong)Size * Size * (ulong)bytesPerPixel; + + using var readback = new VulkanBuffer(context, bytes, + BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + + commands.SubmitAndWait(commandBuffer => + { + textures.TransitionTexture(commandBuffer, texture, ImageLayout.TransferSrcOptimal); + var region = new BufferImageCopy + { + ImageSubresource = new ImageSubresourceLayers(texture.Aspect, 0, 0, 1), + ImageExtent = new Extent3D(Size, Size, 1), + }; + context.Api.CmdCopyImageToBuffer(commandBuffer, texture.Image, + ImageLayout.TransferSrcOptimal, readback.Handle, 1, ®ion); + }); + + var result = new byte[(int)bytes]; + Marshal.Copy(readback.Mapped, result, 0, result.Length); + return result; + } + + private static float ReadHalf(byte[] data, int x, int y, int channel, int bytesPerPixel) + { + int offset = (y * (int)Size + x) * bytesPerPixel + channel * 2; + return (float)BitConverter.ToHalf(data, offset); + } + + private static float ReadByteChannel(byte[] data, int x, int y, int channel) => + data[(y * (int)Size + x) * 4 + channel] / 255f; + + /// + /// Linear interpolation between the two samples of a monotonic-ish + /// array that straddle , returning the + /// fractional index where the crossing happens. + /// + private static float FindThresholdCrossing(float[] values, float threshold) + { + for (int i = 1; i < values.Length; i++) + { + bool crosses = (values[i - 1] < threshold && values[i] >= threshold) + || (values[i - 1] > threshold && values[i] <= threshold); + if (crosses) + { + float denom = values[i] - values[i - 1]; + float t = Math.Abs(denom) > 1e-6f ? (threshold - values[i - 1]) / denom : 0.5f; + return (i - 1) + t; + } + } + throw new InvalidOperationException("no threshold crossing found"); + } + + /// Average red channel over columns [startX, endX) across every row. + private static float AverageRed(byte[] colorBytes, int startX, int endX) + { + float sum = 0f; + int count = 0; + for (int y = 0; y < Size; y++) + for (int x = startX; x < endX; x++) + { + sum += ReadHalf(colorBytes, x, y, 0, 8); + count++; + } + return sum / count; + } +} diff --git a/Optimum.Render.Vulkan.Tests/TemporalFilterTests.cs b/Optimum.Render.Vulkan.Tests/TemporalFilterTests.cs new file mode 100644 index 00000000..91e45d58 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/TemporalFilterTests.cs @@ -0,0 +1,158 @@ +using System; +using System.IO; +using System.Linq; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Shaders; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class TemporalFilterTests(ITestOutputHelper output) +{ + private const int Size = 32; + private const string Triangle = """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); } + """; + + private static int Target(VulkanDevice device, int texture) + { + int target = device.CreateFramebuffer(Size, Size); + device.AttachTexture(target, EnumFramebufferAttachment.ColorAttachment0, texture, 0); + device.SetDrawBuffers(target, 1); return target; + } + + private static unsafe int Texture(VulkanDevice device, Half[]? values = null) + { + fixed (Half* pointer = values) return device.CreateTexture2DRaw(Size, Size, 0x881A, (IntPtr)pointer, 8); + } + + private static void Draw(VulkanDevice device, int program, int target) + { + device.BindFramebuffer(target); device.SetViewport(0, 0, Size, Size); + device.SetDepthTest(false); device.SetDepthMask(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.UseProgram(program); device.DrawFullscreenTriangle(); + } + + [SkippableFact] + public void MotionWriterPreservesReactiveAndRejectsBehindCameraPositions() + { + var device = GpuTest.CreateDevice(output); + try + { + string fragment = "#version 330 core\n" + NativeShaderTree.Read("motion.glsl") + """ + uniform float previousW; + uniform vec2 jitter; + uniform int reactiveOnly; + out vec4 color; + void main() { + vec2 previousNdc = ((gl_FragCoord.xy + vec2(3, -2)) / 32.0) * 2.0 - 1.0; + vec4 previous = vec4(previousNdc * previousW, 0.5 * previousW, previousW); + color = reactiveOnly != 0 ? optimumWriteReactiveOnly(0.7) + : optimumWriteMotion(previous, vec2(32), jitter, 0.3, 0.625); + } + """; + int program = GpuTest.LinkProgram(device, Triangle, fragment, "motion-contract"); + int image = Texture(device), target = Target(device, image); + (float W, float X, float Y, int ReactiveOnly)[] cases = { + (2f, 0f, 0f, 0), (2f, 0.25f, -0.375f, 0), (-1f, 0.25f, -0.375f, 0), + (0f, -0.5f, 0.125f, 0), (1e-6f, 0.25f, -0.375f, 0), (2f, -0.5f, 0.125f, 0), (2f, 0f, 0f, 1), + }; + foreach (var c in cases) + { + device.BeginFrame(); device.BindFramebuffer(target); device.ClearColor(0, 0.75f, 0.75f, 0.75f, 0.75f); + device.SetUniform(program, device.GetUniformLocation(program, "previousW"), c.W); + device.SetUniform(program, device.GetUniformLocation(program, "jitter"), c.X, c.Y); + device.SetUniform(program, device.GetUniformLocation(program, "reactiveOnly"), c.ReactiveOnly); + Draw(device, program, target); + var values = MemoryMarshal.Cast(device.ReadBackLevel0ForTests(image)); + Assert.Equal(Size * Size * 4, values.Length); + bool valid = c.W > 1e-6f && c.ReactiveOnly == 0; + float[] expected = { valid ? 3 + c.X : 0, valid ? -2 + c.Y : 0, + c.ReactiveOnly == 0 ? 0.3f : 0.7f, valid ? 0.625f : 0 }; + for (int i = 0; i < values.Length; i++) + Assert.InRange((float)values[i], expected[i % 4] - 0.002f, expected[i % 4] + 0.002f); + device.Present(); + } + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableTheory] + [InlineData(false)] + [InlineData(true)] + public void SharpenPreservesHdrBypassAndBoundsEdgeAndNoiseAmplification(bool native) + { + Skip.If(ShaderCorpus.AssetRoot == null, "No bootstrapped game assets."); + string root = Path.Combine(Path.GetTempPath(), "optimum-temporal-filter-" + Guid.NewGuid().ToString("N")); + VulkanDevice? device = null; + try + { + if (native) + { + using var compiler = new ShaderCompiler(); + var build = new NativeShaderBuilder(compiler).Build(Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders-vk"), "taa-sharpen"); + Assert.True(build.Success, string.Join("\n", build.Errors)); NativeShaderBuilder.Write(build, root); + } + device = GpuTest.CreateDevice(output, d => { + d.NativeShadersEnabled = native; d.IgnoreModShaderScan = true; + d.NativeShaderDirectory = Path.Combine(root, NativeShaderManifest.DirectoryName); + }); + var source = ShaderCorpus.BuildProgram("taa-sharpen", ShaderCorpus.LoadShaderFiles(), ShaderCorpus.LoadIncludes(), ShaderCorpus.Variants().First()); + var linked = new GpuTest.TestProgram { PassName = "taa-sharpen" }; + foreach (var stage in source) + { + var shader = new GpuTest.TestShader { Type = stage.Stage, Code = stage.Code, PrefixCode = stage.PrefixCode }; + Assert.True(device.CompileShader(shader), device.GetError()); + if (stage.Stage == EnumShaderType.VertexShader) linked.VertexShader = shader; + else if (stage.Stage == EnumShaderType.FragmentShader) linked.FragmentShader = shader; + } + int program = device.LinkProgram(linked); Assert.True(program > 0, device.GetError()); + Assert.Equal(native, device.IsNativeProgram(program)); + device.SetSamplerUnit(program, "inputScene", 0); + device.SetUniform(program, device.GetUniformLocation(program, "inputTexelSize"), 1f / Size, 1f / Size); + int sharpness = device.GetUniformLocation(program, "sharpness"); Assert.True(sharpness >= 0); + int image = Texture(device), target = Target(device, image); + Half[] Pattern(Func value) => Enumerable.Range(0, Size * Size * 4) + .Select(i => (Half)value(i / 4 % Size, i / 4 / Size, i % 4)).ToArray(); + byte[] Run(Half[] values, float strength) + { + int input = Texture(device, values); + device.SetTextureParameter(input, 0x2801, 9729); device.SetTextureParameter(input, 0x2800, 9729); + device.SetTextureParameter(input, 0x2802, 33071); device.SetTextureParameter(input, 0x2803, 33071); + device.BeginFrame(); device.BindTexture(0, input); device.SetUniform(program, sharpness, strength); + Draw(device, program, target); byte[] result = device.ReadBackLevel0ForTests(image); + device.Present(); device.DeleteTexture(input); Assert.Equal(Size * Size * 8, result.Length); return result; + } + float Pixel(byte[] bytes, int x, int y, int channel = 0) => (float)BitConverter.ToHalf(bytes, ((y * Size + x) * 4 + channel) * 2); + var hdr = Pattern((x, y, c) => c switch { 0 => x / (float)Size, 1 => y / (float)Size, + 2 => (x + y) % 8 == 0 ? 3.5f : 0.125f, _ => 0.25f }); + Assert.Equal(MemoryMarshal.AsBytes(hdr.AsSpan()).ToArray(), Run(hdr, 0)); + var edge = Pattern((x, y, c) => c == 3 ? 1 : x < 16 ? 0.2f : 0.8f); + float previousStep = 0; + foreach (float strength in new[] { 0f, 0.5f, 1f }) + { + byte[] pixels = Run(edge, strength); + float dark = Pixel(pixels, 15, 16), bright = Pixel(pixels, 16, 16), step = bright - dark; + if (strength == 0) Assert.InRange(step, 0.595f, 0.605f); + else Assert.True(step > previousStep); + if (strength == 1) { Assert.True(dark < 0.19f); Assert.True(bright > 0.81f); Assert.True(step < 1.2f); } + for (int y = 4; y < Size - 4; y++) + { + Assert.InRange(Pixel(pixels, 4, y), 0.195f, 0.205f); + Assert.InRange(Pixel(pixels, 27, y), 0.795f, 0.805f); + } + previousStep = step; + } + var noise = Pattern((x, y, c) => c == 3 ? 1 : x == 16 && y == 16 ? 0.6f : 0.3f); + byte[] filtered = Run(noise, 1); + Assert.InRange(Pixel(filtered, 16, 16), 0.65f, 0.76f); + Assert.InRange(Pixel(filtered, 17, 16), 0.22f, 0.3f); + } + finally { device?.Dispose(); if (Directory.Exists(root)) Directory.Delete(root, recursive: true); } + GpuTest.AssertClean(device!); + } +} diff --git a/Optimum.Render.Vulkan.Tests/TextureTransferTests.cs b/Optimum.Render.Vulkan.Tests/TextureTransferTests.cs new file mode 100644 index 00000000..9563bc5e --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/TextureTransferTests.cs @@ -0,0 +1,470 @@ +using System; +using System.Linq; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +public class TextureTransferTests(ITestOutputHelper output) +{ + private const string Triangle = """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0, 1); } + """; + + private VulkanDevice Open() => GpuTest.CreateDevice(output); + + private static int Target(VulkanDevice device, int width, int height, params int[] textures) + { + int target = device.CreateFramebuffer(width, height); + for (int i = 0; i < textures.Length; i++) + device.AttachTexture(target, (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + i), textures[i], 0); + device.SetDrawBuffers(target, (1 << textures.Length) - 1); + return target; + } + + private static int Texture(VulkanDevice device, int width, int height) => device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + + private static void Draw(VulkanDevice device, int target, int program, int size) + { + device.BindFramebuffer(target); device.SetViewport(0, 0, size, size); + device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.UseProgram(program); + device.DrawFullscreenTriangle(); + } + + private static void Color(byte[] pixels, byte r, byte g, byte b, byte a = 255) + { + Assert.NotEmpty(pixels); + Assert.Equal(0, pixels.Length % 4); + uint expected = BitConverter.ToUInt32(new byte[] { r, g, b, a }); + Assert.Equal(-1, MemoryMarshal.Cast(pixels).IndexOfAnyExcept(expected)); + } + + private static unsafe byte[] Read(VulkanDevice device, int target, int size) + { + var pixels = new byte[size * size * 4]; + device.BindFramebuffer(target); + fixed (byte* pointer = pixels) device.ReadDefaultFramebuffer(0, 0, size, size, (IntPtr)pointer); + return pixels; + } + + [SkippableFact] + public unsafe void MixedTexelFormatsAndSignedInputRoundTripInOneFrame() + { + var device = Open(); + try + { + byte[] rgba = { 11, 22, 33, 44 }; + float[] floats = Enumerable.Range(0, 16).Select(i => i * 0.25f - 1.5f).ToArray(); + float[] historyDepths = { 12f, 128f }; + int small, wide, historyDepth; + fixed (byte* pointer = rgba) small = device.CreateTexture2D(1, 1, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)pointer, false); + fixed (float* pointer = floats) wide = device.CreateTexture2DRaw(2, 2, 0x8814, (IntPtr)pointer, 16); // RGBA32F + fixed (float* pointer = historyDepths) historyDepth = device.CreateTexture2DRaw(2, 1, 0x822E, (IntPtr)pointer, 4); // R32F + int normalized = device.CreateTexture2DRaw(2, 1, 0x805B, IntPtr.Zero, 8); // RGBA16 + device.UploadTexture2DNormalizedShorts(normalized, 0, 0, 0, 2, 1, + new short[] { short.MinValue, -1, 0, 1, 16384, short.MaxValue, 8192, 0 }); + device.BeginFrame(); + // A four-byte result followed by sixteen-byte texels exercises arena alignment. + Assert.Equal(rgba, device.ReadBackLevel0ForTests(small)); + Assert.Equal(MemoryMarshal.AsBytes(floats.AsSpan()).ToArray(), device.ReadBackLevel0ForTests(wide)); + Assert.Equal(MemoryMarshal.AsBytes(historyDepths.AsSpan()).ToArray(), device.ReadBackLevel0ForTests(historyDepth)); + ushort[] expected = { 0, 0, 0, 2, 32769, 65535, 16384, 0 }; + Assert.Equal(MemoryMarshal.AsBytes(expected.AsSpan()).ToArray(), device.ReadBackLevel0ForTests(normalized)); + Assert.Equal(rgba, device.ReadBackLevel0ForTests(small)); + device.Present(); + ulong oldLifetime = device.TexturesForTests.Get(small)!.Id; + device.DeleteTexture(small); + Assert.Null(device.TexturesForTests.Get(small)); + int reused = Texture(device, 1, 1); + Assert.Equal(small, reused); + Assert.NotEqual(oldLifetime, device.TexturesForTests.Get(reused)!.Id); + Assert.Null(device.TexturesForTests.Get(0)); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void ChangingAnAlreadyBoundSamplerBiasChangesTheNextDraw() + { + using var device = Open(); + int source = device.CreateTexture2D(4, 4, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, IntPtr.Zero, true); + for (int level = 0; level < 3; level++) + { + int side = 4 >> level; + byte[] pixels = new byte[side * side * 4]; + for (int i = 0; i < pixels.Length; i += 4) + { + pixels[i + level] = 255; + pixels[i + 3] = 255; + } + fixed (byte* pointer = pixels) + device.UploadTexture2D(source, level, 0, 0, side, side, + EnumTexturePixelFormat.Rgba, (IntPtr)pointer); + } + + int target = Target(device, 4, 4, Texture(device, 4, 4)); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D source; + out vec4 color; + void main() { color = texture(source, gl_FragCoord.xy / 4.0); } + """, "live-sampler-bias"); + device.SetSamplerUnit(program, "source", 0); + device.BindTexture(0, source); + int sampler = device.CreateSampler(false); + device.BindSampler(0, sampler); + + foreach (int level in new[] { 0, 1, 2, 0 }) + { + device.SetSamplerParameter(sampler, GlEnums.TextureLodBias, (float)level); + device.BeginFrame(); + Draw(device, target, program, 4); + byte[] pixels = Read(device, target, 4); + Color(pixels, level == 0 ? (byte)255 : (byte)0, + level == 1 ? (byte)255 : (byte)0, + level == 2 ? (byte)255 : (byte)0); + device.Present(); + } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void TextureFiltersClampMipSamplingAndSamplerStateIsInterned() + { + var device = Open(); + try + { + int texture = device.CreateTexture2D(8, 8, EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, true); + for (int level = 0; level < 4; level++) + { + int size = 8 >> level; + var pixels = new byte[size * size * 4]; + for (int i = 0; i < pixels.Length; i += 4) + { pixels[i] = (byte)(30 + level * 50); pixels[i + 1] = 60; pixels[i + 2] = 90; pixels[i + 3] = 255; } + fixed (byte* pointer = pixels) device.UploadTexture2D(texture, level, 0, 0, size, size, EnumTexturePixelFormat.Rgba, (IntPtr)pointer); + } + int target = Target(device, 4, 4, Texture(device, 4, 4)); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D source; + out vec4 color; + void main() { color = textureLod(source, vec2(0.5), 3); } + """, "texture-mip-clamp"); + device.SetSamplerUnit(program, "source", 0); + byte[] uniform = Enumerable.Repeat((byte)127, 9 * 5 * 4).ToArray(); + int generated; + fixed (byte* pointer = uniform) generated = device.CreateTexture2D(9, 5, EnumTextureInternalFormat.Rgba8, + EnumTexturePixelFormat.Rgba, (IntPtr)pointer, true); + device.BeginFrame(); + Assert.Equal(4u, device.TexturesForTests.Get(generated)!.MipLevels); + for (uint level = 0; level < 4; level++) + { + byte[] mip = device.ReadBackLevelForTests(generated, level); + Assert.Equal(Math.Max(1, 9 >> (int)level) * Math.Max(1, 5 >> (int)level) * 4, mip.Length); + Assert.Equal(-1, mip.AsSpan().IndexOfAnyExcept((byte)127)); + } + foreach (var state in new (int Filter, int Max, int Red)[] { (0x2600, -1, 30), (0x2700, 1, 80), (0x2700, -1, 180), (0x2601, -1, 30) }) + { + device.SetTextureParameter(texture, GlEnums.TextureMinFilter, state.Filter); + device.SetTextureParameter(texture, GlEnums.TextureMaxLevel, state.Max); + device.BindTexture(0, texture); + Draw(device, target, program, 4); + Color(Read(device, target, 4), (byte)state.Red, 60, 90); + } + device.Present(); + var cache = device.TexturesForTests.Samplers; + var original = device.TexturesForTests.Get(texture)!.State; + var first = cache.Get(original); + Assert.Equal(first.Handle, cache.Get(original).Handle); + Assert.NotEqual(first.Handle, cache.Get(original with { LodBias = -0.5f }).Handle); + foreach (var border in new (float Value, float Alpha, Silk.NET.Vulkan.BorderColor Color)[] { (1f, 1f, Silk.NET.Vulkan.BorderColor.FloatOpaqueWhite), + (0f, 1f, Silk.NET.Vulkan.BorderColor.FloatOpaqueBlack), (0f, 0f, Silk.NET.Vulkan.BorderColor.FloatTransparentBlack) }) + { + device.SetTextureBorderColor(texture, border.Value, border.Value, border.Value, border.Alpha); + Assert.Equal(border.Color, device.TexturesForTests.Get(texture)!.State.BorderColor); + } + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void PartialReadbackPreservesUniformSnapshotsAndFrameIdentity() + { + var device = Open(); + try + { + int targetA = Target(device, 4, 4, Texture(device, 4, 4)), targetB = Target(device, 4, 4, Texture(device, 4, 4)); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + layout(std140) uniform Tint { vec4 tint; }; + out vec4 color; + void main() { color = tint; } + """, "readback-uniform-snapshot"); + int ubo = device.CreateUniformBuffer(program, 0, "Tint", 16); + device.BindUniformBuffer(ubo); + void Tint(byte red) + { + float[] data = { red / 255f, 60 / 255f, 90 / 255f, 1 }; + fixed (float* pointer = data) device.UpdateUniformBuffer(ubo, (IntPtr)pointer, 0, 16); + } + device.BeginFrame(); device.Present(); + long pacing = VulkanStats.WaitCount(WaitSite.FramePacing), idle = VulkanStats.WaitCount(WaitSite.DeviceWaitIdle); + long flush = VulkanStats.WaitCount(WaitSite.FlushFrame), reads = VulkanStats.WaitCount(WaitSite.Readback); + long submits = VulkanStats.WaitCount(WaitSite.QueueSubmit); + for (int frame = 0; frame < 8; frame++) + { + device.BeginFrame(); + ulong frameId = device.LatencyFrameId; + Tint((byte)(20 + frame)); Draw(device, targetA, program, 4); + Color(Read(device, targetA, 4), (byte)(20 + frame), 60, 90); + Assert.Equal(frameId, device.LatencyFrameId); + // The unchanged block's snapshot must remain live across the partial submission. + Draw(device, targetB, program, 4); + Tint((byte)(100 + frame)); Draw(device, targetA, program, 4); + device.Present(); + device.BeginFrame(); + Color(Read(device, targetA, 4), (byte)(100 + frame), 60, 90); + Color(Read(device, targetB, 4), (byte)(20 + frame), 60, 90); + device.Present(); + } + Assert.Equal(16, VulkanStats.WaitCount(WaitSite.FramePacing) - pacing); + Assert.Equal(24, VulkanStats.WaitCount(WaitSite.Readback) - reads); + Assert.Equal(40, VulkanStats.WaitCount(WaitSite.QueueSubmit) - submits); + Assert.Equal(idle, VulkanStats.WaitCount(WaitSite.DeviceWaitIdle)); + Assert.Equal(flush, VulkanStats.WaitCount(WaitSite.FlushFrame)); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void ReadbackArenaGrowthKeepsEveryPixelAndClearInOrder() + { + var device = Open(); + try + { + const int size = 1024; + int target = Target(device, size, size, Texture(device, size, size)); + device.BeginFrame(); + for (int i = 0; i < 3; i++) + { + device.BindFramebuffer(target); + device.ClearColor(0, (30 + i * 60) / 255f, 60 / 255f, 90 / 255f, 1); + Color(Read(device, target, size), (byte)(30 + i * 60), 60, 90); + } + device.Present(); + for (int i = 0; i < 3; i++) { device.BeginFrame(); device.Present(); } + device.BeginFrame(); + Color(Read(device, target, size), 150, 60, 90); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableTheory] + [InlineData(1)] + [InlineData(17)] + [InlineData(31)] + public unsafe void SparseFragmentOutputsOnlyChangeEnabledAttachments(int mask) + { + var device = Open(); + try + { + int[] textures = Enumerable.Range(0, 5).Select(_ => Texture(device, 4, 4)).ToArray(); + int target = Target(device, 4, 4, textures); + int program = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + layout(location = 0) out vec4 color; + layout(location = 4) out vec4 motion; + void main() { color = vec4(1, 0, 0, 1); motion = vec4(0, 1, 0, 1); } + """, "sparse-attachments"); + device.BeginFrame(); device.BindFramebuffer(target); + for (int i = 0; i < textures.Length; i++) device.ClearColor(i, (20 + i * 20) / 255f, 60 / 255f, 90 / 255f, 1); + device.SetDrawBuffers(target, mask); + Draw(device, target, program, 4); + for (int i = 0; i < textures.Length; i++) + { + byte[] pixels = device.ReadBackLevel0ForTests(textures[i]); + if (i == 0) Color(pixels, 255, 0, 0); + else if (i == 4 && (mask & 16) != 0) Color(pixels, 0, 255, 0); + else Color(pixels, (byte)(20 + i * 20), 60, 90); + } + int blend = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + layout(location = 0) out vec4 color; + layout(location = 4) out vec4 motion; + void main() { color = vec4(0.8, 0.8, 0.8, 0); motion = color; } + """, "independent-attachment-blend"); + device.SetDrawBuffers(target, 17); device.BindFramebuffer(target); device.UseProgram(blend); + device.DeclarePass(new Optimum.Render.Vulkan.Graph.PassDeclaration { + Name = "sparse-compose", FramebufferId = target, ColorSlots = 17, + }); + device.SetBlend(true, EnumBlendMode.Standard); + device.SetBlendFuncSeparate(4, 1, 0, 1, 0); + device.DrawFullscreenTriangle(); + Color(device.ReadBackLevel0ForTests(textures[0]), 255, 0, 0); + Color(device.ReadBackLevel0ForTests(textures[4]), 204, 204, 204, 0); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + + [SkippableFact] + public unsafe void ThreeOitOutputsReachThreeDifferentArrayLayers() + { + using var device = Open(); + int layers = device.CreateTexture2DArray(4, 4, 3, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba); + int accumulation = device.CreateFramebuffer(4, 4); + for (int layer = 0; layer < 3; layer++) + device.AttachTexture(accumulation, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + 3 + layer), + layers, layer); + device.SetDrawBuffers(accumulation, 0x38); + int write = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + layout(location = 3) out vec4 red; + layout(location = 4) out vec4 green; + layout(location = 5) out vec4 blue; + void main() { + red = vec4(1, 0, 0, 1); + green = vec4(0, 1, 0, 1); + blue = vec4(0, 0, 1, 1); + } + """, "oit-array-write"); + int inspect = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2DArray source; + out vec4 color; + void main() { + int layer = gl_FragCoord.x < 1.0 ? 0 : (gl_FragCoord.x < 2.0 ? 1 : 2); + color = texelFetch(source, ivec3(0, 0, layer), 0); + } + """, "oit-array-inspect"); + int result = Target(device, 3, 1, Texture(device, 3, 1)); + device.SetSamplerUnit(inspect, "source", 0); + device.BeginFrame(); + Draw(device, accumulation, write, 4); + device.BindTexture(0, layers); + device.BindFramebuffer(result); device.SetViewport(0, 0, 3, 1); + device.SetDepthTest(false); device.SetCullFace(false); + device.SetBlend(false, EnumBlendMode.Standard); device.UseProgram(inspect); + device.DrawFullscreenTriangle(); + var pixels = new byte[12]; + fixed (byte* pointer = pixels) device.ReadDefaultFramebuffer(0, 0, 3, 1, (IntPtr)pointer); + Assert.Equal(new byte[] { 255, 0, 0, 255, 0, 255, 0, 255, 0, 0, 255, 255 }, pixels); + device.Present(); + GpuTest.AssertClean(device); + } + + [SkippableFact] + public void BoundDepthCanBeSampledAfterAWriteWithoutChangingItsContents() + { + var device = Open(); + try + { + int color = Texture(device, 4, 4), depth = device.CreateTexture2DRaw(4, 4, 0x8CAC, IntPtr.Zero, 4); + int target = Target(device, 4, 4, color); + device.AttachTexture(target, EnumFramebufferAttachment.DepthAttachment, depth, 0); + int write = GpuTest.LinkProgram(device, """ + #version 330 core + void main() { gl_Position = vec4(-1 + ((gl_VertexID & 1) << 2), -1 + ((gl_VertexID & 2) << 1), 0.5, 1); } + """, """ + #version 330 core + out vec4 color; + void main() { color = vec4(1); } + """, "depth-write"); + int sample = GpuTest.LinkProgram(device, Triangle, """ + #version 330 core + uniform sampler2D depthTex; + out vec4 color; + void main() { color = vec4(texelFetch(depthTex, ivec2(gl_FragCoord.xy), 0).r, 0, 0, 1); } + """, "depth-read-only"); + device.SetSamplerUnit(sample, "depthTex", 0); + device.BeginFrame(); device.BindFramebuffer(target); device.SetViewport(0, 0, 4, 4); + device.SetCullFace(false); device.SetBlend(false, EnumBlendMode.Standard); + device.SetDepthTest(true); device.SetDepthMask(true); device.SetDepthFunc(0x207); + device.ClearDepth(1); device.UseProgram(write); device.DrawFullscreenTriangle(); + device.SetDepthMask(false); device.BindTexture(0, depth); device.UseProgram(sample); + device.DrawFullscreenTriangle(); + Color(Read(device, target, 4), 191, 0, 0); + float[] values = MemoryMarshal.Cast(device.ReadBackLevel0ForTests(depth)).ToArray(); + Assert.All(values, value => Assert.Equal(0.75f, value)); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + [Fact] + public unsafe void PoisonFillsPartialHostWordsAndRequiresAnEnabledSetting() + { + foreach (string? disabled in new string?[] { null, "", "0" }) Assert.False(VulkanContext.PoisonRequested(disabled)); + foreach (string enabled in new[] { "1", " 1 " }) Assert.True(VulkanContext.PoisonRequested(enabled)); + var bytes = new byte[11]; + fixed (byte* pointer = bytes) VulkanPoison.FillHostMemory((IntPtr)pointer, (ulong)bytes.Length); + Assert.Equal(new byte[] { 0xEF, 0xBE, 0xAD, 0xDE, 0xEF, 0xBE, 0xAD, 0xDE, 0xEF, 0xBE, 0xAD }, bytes); + } + + [SkippableFact] + public unsafe void PoisonSurvivesAttachmentLoadsAndIsReplacedByClears() + { + var device = GpuTest.CreateDevice(output, d => { + var configure = d.ConfigureContextOptions; + d.ConfigureContextOptions = options => { configure?.Invoke(options); options.Poison = true; }; + }); + try + { + Assert.True(device.ContextForTests.PoisonFreshResources); + using (var buffer = new VulkanBuffer(device.ContextForTests, 68, BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit)) + { + var words = new ReadOnlySpan((void*)buffer.Mapped, 17); + Assert.Equal(-1, words.IndexOfAnyExcept(0xDEADBEEFu)); + } + Format[] formats = { Format.R8G8B8A8Unorm, Format.R8G8B8A8Srgb, Format.R16G16B16A16Sfloat, + Format.R32Sfloat, Format.R32Uint, Format.D32Sfloat }; + int[] textures = formats.Select(format => device.TexturesForTests.Create(8, 8, format)).ToArray(); + int target = Target(device, 8, 8, textures[..5]); + device.AttachTexture(target, EnumFramebufferAttachment.DepthAttachment, textures[5], 0); + int program = GpuTest.LinkProgram(device, Triangle, "#version 330 core\nvoid main() {}", "poison-load"); + device.BeginFrame(); device.SetDepthMask(false); Draw(device, target, program, 8); + for (int kind = 0; kind < textures.Length; kind++) + { + byte[] bytes = device.ReadBackLevel0ForTests(textures[kind]); + Assert.Equal(8 * 8 * (kind == 2 ? 8 : 4), bytes.Length); + if (kind < 2) { Color(bytes, 255, 0, 255); continue; } + if (kind == 2) + { + foreach (Half value in MemoryMarshal.Cast(bytes)) Assert.True(Half.IsNaN(value)); + } + else if (kind == 3) + { + foreach (float value in MemoryMarshal.Cast(bytes)) Assert.True(float.IsNaN(value)); + } + else if (kind == 4) Assert.Equal(-1, MemoryMarshal.Cast(bytes).IndexOfAnyExcept(0xDEADBEEFu)); + else Assert.Equal(-1, MemoryMarshal.Cast(bytes).IndexOfAnyExcept(0.5f)); + } + device.BindFramebuffer(target); device.SetDepthMask(true); + device.ClearColor(0, 0.25f, 0.2f, 0.75f, 1); device.ClearDepth(1); + Color(device.ReadBackLevel0ForTests(textures[0]), 64, 51, 191); + Assert.Equal(-1, MemoryMarshal.Cast(device.ReadBackLevel0ForTests(textures[5])).IndexOfAnyExcept(1f)); + device.Present(); + } + finally { device.Dispose(); } + GpuTest.AssertClean(device); + } + +} diff --git a/Optimum.Render.Vulkan.Tests/UiSeparationTests.cs b/Optimum.Render.Vulkan.Tests/UiSeparationTests.cs new file mode 100644 index 00000000..e24fe114 --- /dev/null +++ b/Optimum.Render.Vulkan.Tests/UiSeparationTests.cs @@ -0,0 +1,453 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Reflection; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Optimum.Render.Vulkan.Platform; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client.NoObf; +using Xunit; +using Xunit.Abstractions; + +namespace Optimum.Render.Vulkan.Tests; + +/// +/// World/UI separation on Vulkan (VulkanClientPlatform.UiSeparation.cs): the frame is rendered +/// HUD-less, the GUI lands in its own image with real coverage, and the compose puts it back. +/// +/// Driven through the real platform bodies - the scope's open, the stated GUI-shaped draws that +/// only name Default, the compose, the snapshot - on a headless device, whose Default target is +/// the window-sized image a headless run reads back. Every judged frame follows unjudged ones with +/// EndFrame between them and no readback in the loop, so an image that is really an earlier +/// frame's fails here. Validation runs with sync and best practices on, as in every GPU test. +/// +public class UiSeparationTests(ITestOutputHelper output) +{ + private const int Size = 16; + private const int Half = Size / 2; + + // The scene under the UI, and two straight-alpha UI layers: one over the whole image at 0.4, + // one over the left half at 0.6. Distinguishable in every channel. + private static readonly double[] Scene = { 40, 90, 200, 255 }; + private static readonly double[] LayerA = { 200, 100, 50, 0.4 }; + private static readonly double[] LayerB = { 20, 180, 240, 0.6 }; + + private const string FullscreenVertex = """ + #version 330 core + void main(void) + { + float x = -1.0 + float((gl_VertexID & 1) << 2); + float y = -1.0 + float((gl_VertexID & 2) << 1); + gl_Position = vec4(x, y, 0.0, 1.0); + } + """; + + private static readonly string LayerAFragment = """ + #version 330 core + out vec4 outColor; + void main(void) { outColor = vec4(200.0 / 255.0, 100.0 / 255.0, 50.0 / 255.0, 0.4); } + """; + + private static readonly string LayerBFragment = """ + #version 330 core + out vec4 outColor; + void main(void) + { + if (gl_FragCoord.x >= 8.0) discard; + outColor = vec4(20.0 / 255.0, 180.0 / 255.0, 240.0 / 255.0, 0.6); + } + """; + + // The shipped sources/shaders/ui-compose.{vsh,fsh}, read from the tree so the test links + // exactly what ships (its native twin is pinned by NativeShaderParityTests). + private static string ComposeSource(string extension) => + File.ReadAllText(Path.Combine(ShaderCorpus.RepositoryRoot, "sources", "shaders", "ui-compose." + extension)); + + private sealed class SeparationPlatform : VulkanClientPlatform + { + public SeparationPlatform() : base(null!) + { + } + + public override Size2i OptimumWindowClientSize() => new(Size, Size); + } + + /// + /// The three images at their two moments. Before the compose the window image holds the scene + /// and nothing else; the UI image holds exactly the GUI - the over-operator's premultiplied + /// colour and coverage where it drew, transparent black where it did not, including where an + /// earlier frame drew; after the compose the window image is ui.rgb + scene * (1 - ui.a), which + /// is what the same layers drawn straight onto the scene give. + /// + [SkippableFact] + public void TheGuiLandsInItsOwnImageAndIsComposedOverTheScene() + { + using Session session = Open(); + SeparationPlatform platform = session.Platform; + VulkanDevice seam = session.Seam; + FrameBufferRef ui = platform.UiTargetFrameBuffer!; + Assert.Equal(VulkanClientPlatform.OptimumUiTargetIndex, platform.UiTargetFrameBufferIndex); + Assert.Equal(Size, ui.Width); + + for (int frame = 0; frame < 3; frame++) + { + session.RunFrame(layerA: true, layerB: true, read: false); + } + + // Both layers: coverage accumulates as the over-operator, not as alpha squared. + Frame both = session.RunFrame(layerA: true, layerB: true, read: true); + double[] uiRight = Premultiplied(LayerA); + double[] uiLeft = Over(Premultiplied(LayerB), uiRight); + AssertHalf(both.Ui, left: false, uiRight, "UI image, layer A alone"); + AssertHalf(both.Ui, left: true, uiLeft, "UI image, layer B over layer A"); + AssertHalf(both.Before, left: true, Scene, "window before the compose"); + AssertHalf(both.Before, left: false, Scene, "window before the compose"); + AssertHalf(both.After, left: false, Composed(uiRight), "window after the compose"); + AssertHalf(both.After, left: true, Composed(uiLeft), "window after the compose"); + + // Only layer B: the right half of the UI image is empty again, and the window shows the + // scene there - the image is cleared every frame, not accumulated across frames. + Frame onlyB = session.RunFrame(layerA: false, layerB: true, read: true); + AssertHalf(onlyB.Ui, left: false, new double[] { 0, 0, 0, 0 }, "UI image, nothing drawn"); + AssertHalf(onlyB.Ui, left: true, Premultiplied(LayerB), "UI image, layer B alone"); + AssertHalf(onlyB.After, left: false, Scene, "window where no UI drew"); + AssertHalf(onlyB.After, left: true, Composed(Premultiplied(LayerB)), "window after the compose"); + + GpuTest.AssertClean(seam); + } + + /// + /// The scope is a window of the frame and nothing more. Open, Default resolves to the UI image + /// and Standard takes over-operator alpha; the compose closes both, and a second compose in the + /// frame records nothing. A scope nobody composed (a GUI renderer that threw) is closed by the + /// next BeginFrame, so the world pass after it blends with vanilla's factors: one 0.4 layer over + /// transparent black keeps 0.4 * 0.4 = 0.16 coverage, not 0.4. + /// + [SkippableFact] + public void TheScopeClosesAtTheComposeAndAtTheNextFrame() + { + using Session session = Open(); + SeparationPlatform platform = session.Platform; + VulkanDevice seam = session.Seam; + FrameBufferRef ui = platform.UiTargetFrameBuffer!; + + platform.BeginFrame(); + platform.OpenUiScope(); + Assert.True(platform.UiScopeOpen); + Assert.Equal(ui.FboId, seam.DefaultFramebufferRedirect); + platform.OptimumComposeUiTarget(); + Assert.False(platform.UiScopeOpen); + Assert.Equal(0, seam.DefaultFramebufferRedirect); + long passes = seam.NativePassesForTests; + platform.OptimumComposeUiTarget(); + Assert.Equal(passes, seam.NativePassesForTests); + platform.EndFrame(); + + // Left open, as by a GUI renderer that threw past both compose call sites. + platform.BeginFrame(); + platform.OpenUiScope(); + platform.EndFrame(); + + platform.BeginFrame(); + Assert.False(platform.UiScopeOpen); + Assert.Equal(0, seam.DefaultFramebufferRedirect); + + // A world-shaped Standard draw into a target cleared to transparent black. + FrameBufferRef world = platform.SceneNoHudFrameBuffer!; + seam.ClearNativeColor(world.FboId, 0, 0f, 0f, 0f, 0f); + platform.CurrentFrameBuffer = world; + session.DrawLayer(session.LayerAProgram); + byte[] pixels = seam.ReadBackLevel0ForTests(world.ColorTextureIds[0]); + platform.CurrentFrameBuffer = null; + platform.EndFrame(); + + int alpha = pixels[(Half * Size + Half) * 4 + 3]; + output.WriteLine("world-pass coverage after an uncomposed scope: " + alpha + " (vanilla 41, scoped 102)"); + Assert.InRange(alpha, 39, 43); + GpuTest.AssertClean(seam); + } + + /// + /// The snapshot is the composited scene texel for texel: Primary colour 0 copied into slot 23 + /// at the end of the composition, and flagged as captured for that frame only. + /// + [SkippableFact] + public void TheSnapshotIsTheCompositedScene() + { + using Session session = Open(); + SeparationPlatform platform = session.Platform; + VulkanDevice seam = session.Seam; + FrameBufferRef primary = platform.FrameBuffers[0]; + FrameBufferRef snapshot = platform.SceneNoHudFrameBuffer!; + Assert.Equal(VulkanClientPlatform.OptimumSceneNoHudIndex, platform.SceneNoHudFrameBufferIndex); + + byte[] captured = Array.Empty(); + for (int frame = 0; frame < 4; frame++) + { + platform.BeginFrame(); + // A different scene every frame, so a copy of an earlier one cannot pass. + float phase = frame / 4f; + seam.ClearNativeColor(primary.FboId, 0, 0.1f + phase * 0.5f, 0.7f - phase * 0.4f, 0.3f, 1f); + platform.CaptureSceneNoHud(); + Assert.True(platform.SceneNoHudCaptured); + if (frame == 3) captured = seam.ReadBackLevel0ForTests(snapshot.ColorTextureIds[0]); + platform.EndFrame(); + } + + var expected = new double[] { (0.1 + 0.75 * 0.5) * 255, (0.7 - 0.75 * 0.4) * 255, 0.3 * 255, 255 }; + AssertHalf(captured, left: true, expected, "snapshot"); + AssertHalf(captured, left: false, expected, "snapshot"); + GpuTest.AssertClean(seam); + } + + /// + /// The factor rule without a device: only Standard's exact factor set changes, only its source + /// alpha factor, and only for a draw into the UI image while the scope is open. + /// + [Fact] + public void OnlyStandardDrawnIntoTheUiImageTakesOverOperatorAlpha() + { + AttachmentBlend standard = AttachmentBlend.For(true, EnumBlendMode.Standard).ForUiImage(); + Assert.Equal(BlendFactor.SrcAlpha, standard.SrcColor); + Assert.Equal(BlendFactor.OneMinusSrcAlpha, standard.DstColor); + Assert.Equal(BlendFactor.One, standard.SrcAlpha); + Assert.Equal(BlendFactor.OneMinusSrcAlpha, standard.DstAlpha); + + foreach (EnumBlendMode mode in new[] + { + EnumBlendMode.PremultipliedAlpha, EnumBlendMode.Brighten, EnumBlendMode.Multiply, + EnumBlendMode.Glow, EnumBlendMode.Overlay, + }) + { + AttachmentBlend plain = AttachmentBlend.For(true, mode); + Assert.Equal(plain, plain.ForUiImage()); + } + + var stated = new StatedRenderState(); + stated.SetBlendEnabled(true); + stated.SetBlendMode(EnumBlendMode.Standard); + Assert.Equal(BlendFactor.SrcAlpha, stated.AttachmentFor(PassDeclaration.DefaultFramebuffer, 0).SrcAlpha); + + stated.UiImageFramebuffer = 42; + Assert.Equal(BlendFactor.One, stated.AttachmentFor(PassDeclaration.DefaultFramebuffer, 0).SrcAlpha); + Assert.Equal(BlendFactor.One, stated.AttachmentFor(42, 0).SrcAlpha); + Assert.Equal(BlendFactor.SrcAlpha, stated.AttachmentFor(7, 0).SrcAlpha); + + stated.UiImageFramebuffer = 0; + Assert.Equal(BlendFactor.SrcAlpha, stated.AttachmentFor(PassDeclaration.DefaultFramebuffer, 0).SrcAlpha); + } + + [Fact] + public void StatedBlendSnapshotsAreReusedUntilARelevantStateChanges() + { + var stated = new StatedRenderState(); + AttachmentBlend[] first = stated.BlendFor(7, 2); + Assert.Same(first, stated.BlendFor(7, 2)); + + stated.SetDrawBuffers(7, 0b11); + AttachmentBlend[] bothSlots = stated.BlendFor(7, 2); + Assert.NotSame(first, bothSlots); + Assert.NotEqual(first[1].WriteMask, bothSlots[1].WriteMask); + + stated.SetBlendEnabled(true); + stated.SetBlendMode(EnumBlendMode.Standard); + AttachmentBlend[] standard = stated.BlendFor(PassDeclaration.DefaultFramebuffer, 1); + stated.UiImageFramebuffer = 42; + AttachmentBlend[] ui = stated.BlendFor(PassDeclaration.DefaultFramebuffer, 1); + Assert.NotSame(standard, ui); + Assert.Equal(BlendFactor.SrcAlpha, standard[0].SrcAlpha); + Assert.Equal(BlendFactor.One, ui[0].SrcAlpha); + } + + // ------------------------------------------------------------------------ arithmetic + + /// A straight-alpha layer (rgb in bytes, alpha 0..1) as premultiplied bytes. + private static double[] Premultiplied(double[] layer) => + new[] { layer[0] * layer[3], layer[1] * layer[3], layer[2] * layer[3], layer[3] * 255 }; + + /// The over-operator on premultiplied bytes. + private static double[] Over(double[] top, double[] under) + { + double keep = 1 - top[3] / 255; + return new[] { top[0] + under[0] * keep, top[1] + under[1] * keep, top[2] + under[2] * keep, top[3] + under[3] * keep }; + } + + /// The window after the compose: the UI over the opaque scene. + private static double[] Composed(double[] ui) => Over(ui, Scene); + + private void AssertHalf(byte[] pixels, bool left, double[] expected, string what) + { + int wrong = 0; + string first = ""; + for (int y = 0; y < Size; y++) + { + for (int x = left ? 0 : Half; x < (left ? Half : Size); x++) + { + int i = (y * Size + x) * 4; + for (int c = 0; c < 4; c++) + { + if (Math.Abs(pixels[i + c] - expected[c]) <= 2) continue; + if (wrong == 0) + { + first = $" first at ({x},{y}): {pixels[i]},{pixels[i + 1]},{pixels[i + 2]},{pixels[i + 3]}"; + } + wrong++; + break; + } + } + } + output.WriteLine($"{what} ({(left ? "left" : "right")}): expected " + + $"{expected[0]:F0},{expected[1]:F0},{expected[2]:F0},{expected[3]:F0}, wrong {wrong}/{Half * Size}{first}"); + Assert.True(wrong == 0, what + ": " + wrong + " pixels off by more than 2/255." + first); + } + + // ------------------------------------------------------------------------ driving + + private readonly record struct Frame(byte[] Ui, byte[] Before, byte[] After); + + private Session Open() + { + Session? session = Session.TryOpen(output); + Skip.If(session == null, "No usable Vulkan device."); + return session!; + } + + private sealed class Session : IDisposable + { + public SeparationPlatform Platform { get; private init; } = null!; + public VulkanDevice Seam => Platform.GraphicsDevice!; + public int LayerAProgram { get; private set; } + public int LayerBProgram { get; private set; } + + private ShaderProgram? previousCompose; + private string dataPath = ""; + + public static Session? TryOpen(ITestOutputHelper output) + { + string dataPath = Path.Combine(Path.GetTempPath(), "optimum-ui-separation-" + Guid.NewGuid().ToString("N")); + var platform = new SeparationPlatform + { + DeviceFactory = GpuTest.NewDevice, + CrashMarkerDataPath = dataPath, + }; + if (!platform.InitializeGraphics(IntPtr.Zero, Size, Size, out string reason)) + { + output.WriteLine("Vulkan unavailable: " + reason); + platform.ShutdownGraphics(); + return null; + } + + var session = new Session + { + Platform = platform, + dataPath = dataPath, + previousCompose = ShaderPrograms.UiCompose, + }; + VulkanDevice seam = platform.GraphicsDevice!; + + // Primary as the composition leaves it, plus the two separation images, installed + // the way SetupDefaultFrameBuffers installs them. + var list = new List(); + for (int i = 0; i <= 24; i++) list.Add(null!); + list[0] = ColorTarget(seam); + platform.AllocateUiSeparationTargets(list, Size, Size); + const BindingFlags flags = BindingFlags.Instance | BindingFlags.NonPublic; + typeof(ClientPlatformWindows).GetField("frameBuffers", flags)!.SetValue(platform, list); + + session.LayerAProgram = GpuTest.LinkProgram(seam, FullscreenVertex, LayerAFragment, "ui-layer-a"); + session.LayerBProgram = GpuTest.LinkProgram(seam, FullscreenVertex, LayerBFragment, "ui-layer-b"); + int compose = GpuTest.LinkProgram( + seam, ComposeSource("vsh"), ComposeSource("fsh"), "ui-compose"); + ShaderPrograms.UiCompose = new ShaderProgram { ProgramId = compose, PassName = "ui-compose" }; + return session; + } + + /// + /// One frame from the blit on: the window image painted with the scene, the scope opened + /// as the blit's end opens it, the GUI-shaped draws naming only Default, and the compose. + /// + public Frame RunFrame(bool layerA, bool layerB, bool read) + { + VulkanDevice seam = Seam; + FrameBufferRef ui = Platform.UiTargetFrameBuffer!; + Platform.BeginFrame(); + seam.ClearNativeColor(PassDeclaration.DefaultFramebuffer, 0, + (float)(Scene[0] / 255), (float)(Scene[1] / 255), (float)(Scene[2] / 255), 1f); + + Platform.OpenUiScope(); + Platform.CurrentFrameBuffer = null; + if (layerA) DrawLayer(LayerAProgram); + if (layerB) DrawLayer(LayerBProgram); + + byte[] uiPixels = read ? seam.ReadBackLevel0ForTests(ui.ColorTextureIds[0]) : Array.Empty(); + byte[] before = read ? ReadWindow(seam) : Array.Empty(); + Platform.OptimumComposeUiTarget(); + byte[] after = read ? ReadWindow(seam) : Array.Empty(); + Platform.EndFrame(); + return new Frame(uiPixels, before, after); + } + + /// A straight-alpha fullscreen layer under Standard, into whatever is bound. + public void DrawLayer(int program) + { + Platform.GlViewport(0, 0, Size, Size); + Platform.GlDisableDepthTest(); + Platform.GlDepthMask(false); + Platform.GlDisableCullFace(); + Platform.GlToggleBlend(true, EnumBlendMode.Standard); + Platform.UseShaderProgram(program); + Platform.RenderFullscreenTriangle(null!); + Platform.UseShaderProgram(0); + } + + /// The window image itself, whatever Default resolves to right now, in RGBA. + private static unsafe byte[] ReadWindow(VulkanDevice seam) + { + var pixels = new byte[Size * Size * 4]; + fixed (byte* destination = pixels) + { + seam.ReadFramebufferColor(seam.DefaultFramebufferId, 0, 0, Size, Size, (IntPtr)destination); + } + if (seam.DefaultColorFormat == Format.B8G8R8A8Unorm || seam.DefaultColorFormat == Format.B8G8R8A8Srgb) + { + for (int i = 0; i < pixels.Length; i += 4) (pixels[i], pixels[i + 2]) = (pixels[i + 2], pixels[i]); + } + return pixels; + } + + private static FrameBufferRef ColorTarget(VulkanDevice seam) + { + var target = new FrameBufferRef + { + Width = Size, + Height = Size, + FboId = seam.CreateFramebuffer(Size, Size), + ColorTextureIds = new[] + { + seam.CreateTexture2D(Size, Size, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false), + }, + }; + seam.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + seam.SetDrawBuffers(target.FboId, 1); + Assert.True(seam.CheckFramebufferComplete(target.FboId, out string status), status); + return target; + } + + public void Dispose() + { + ShaderPrograms.UiCompose = previousCompose; + Platform.ShutdownGraphics(); + try + { + Directory.Delete(dataPath, true); + } + catch (DirectoryNotFoundException) + { + } + } + } +} diff --git a/Optimum.Render.Vulkan/AmbientOcclusion/GtaoRenderer.cs b/Optimum.Render.Vulkan/AmbientOcclusion/GtaoRenderer.cs new file mode 100644 index 00000000..9707912b --- /dev/null +++ b/Optimum.Render.Vulkan/AmbientOcclusion/GtaoRenderer.cs @@ -0,0 +1,387 @@ +using System; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using System.Collections.Concurrent; +using System.IO; +using System.Reflection; +using System.Text; +using System.Text.RegularExpressions; + +namespace Optimum.Render.Vulkan.AmbientOcclusion; + +/// +/// The AO pipeline on the device (docs/vulkan.md#ambient-occlusion section C): prefilter +/// (Primary depth to five R32F levels), main (visibility term and packed edges) and the +/// edge-aware denoise, as frame-graph compute passes. Owns its targets, sized to the depth +/// it is given, and its programs, compiled for the storage formats the device chose. +/// +/// The output is the visibility at render resolution, unscaled from the 1.5 packing, with +/// sky and hand-view pixels exactly 1; the compose pass (scene-ssao, OPTIMUMAO) multiplies +/// it into Primary colour 0 before the TAA resolve and applies the water, fog and OIT +/// attenuation there, so this term stays a pure visibility for measurement. +/// +internal sealed class GtaoRenderer : IDisposable +{ + public const int DepthLevels = 5; + + /// The smallest depth the five-level chain covers. + public const int MinimumExtent = 16; + + private readonly VulkanDevice _device; + private int _hilbert; + private int _workingDepth; + private int _workingTerm; + private int _edges; + private int _output; + private int _scratchA; + private int _scratchB; + private uint _width; + private uint _height; + private int _prefilter; + private int _main; + private int _denoise; + private Format _programDepthFormat; + private Format _programTermFormat; + + public GtaoRenderer(VulkanDevice device) => _device = device; + + /// Why the last returned 0. + public string? LastError { get; private set; } + + /// True once the programs failed to compile: a device condition, not a frame's. + public bool ProgramsFailed { get; private set; } + + /// Working depth, five levels of view-space depth (the debug output "mip 0" is level 0). + public int WorkingDepthTexture => _workingDepth; + + /// The pre-denoise term, visibility / 1.5. + public int WorkingTermTexture => _workingTerm; + + /// The packed 2-bit edges. + public int EdgesTexture => _edges; + + /// The denoised visibility. + public int OutputTexture => _output; + + public int HilbertTexture => _hilbert; + + /// + /// Records the three passes into the open frame and returns , + /// or 0 with set when nothing could be recorded. + /// + /// Primary's D32 depth. + /// Primary colour 2 (gNormal): GL view-space normal, class in w. + /// The column-major GL projection the G-buffer was drawn with. + /// Preset, variants and uniforms. + /// The frame index while TAA accumulates, 0 otherwise. + public int Render(int depthTexture, int normalTexture, float[] projection, GtaoSettings settings, uint noiseIndex) + { + LastError = null; + GtaoProjection? reconstruction = GtaoProjection.From(projection); + if (reconstruction == null) return Fail("the projection is not a GL perspective matrix"); + + VulkanTexture? depth = _device.TextureOf(depthTexture); + VulkanTexture? normal = _device.TextureOf(normalTexture); + if (depth == null || normal == null) return Fail("depth or normal texture missing"); + if (depth.Width != normal.Width || depth.Height != normal.Height) return Fail("depth and normal sizes differ"); + if (depth.Width < MinimumExtent || depth.Height < MinimumExtent) return Fail("the target is smaller than 16x16"); + + if (!EnsureTargets(depth.Width, depth.Height)) return Fail("storage targets could not be created"); + if (!EnsurePrograms()) return 0; + + byte[] push = settings.PushConstants(reconstruction.Value, noiseIndex); + + uint blocksX = (depth.Width + 15) / 16; + uint blocksY = (depth.Height + 15) / 16; + var prefilterBindings = new ComputeBinding[1 + DepthLevels]; + prefilterBindings[0] = new ComputeBinding(0, depthTexture, ComputeAccess.Sampled); + for (uint level = 0; level < DepthLevels; level++) + { + prefilterBindings[1 + level] = new ComputeBinding(1 + level, _workingDepth, ComputeAccess.StorageWrite, BaseMip: level); + } + if (!_device.RecordComputePass(new ComputePassDeclaration + { + Name = "gtao-prefilter", + ProgramId = _prefilter, + Bindings = prefilterBindings, + Dispatches = new[] { ComputeDispatch.Explicit((blocksX + 7) / 8, (blocksY + 7) / 8, 1, push) }, + })) return Fail(_device.GetError()); + + if (!_device.RecordComputePass(new ComputePassDeclaration + { + Name = "gtao-main", + ProgramId = _main, + Specialization = settings.MainSpecialization(), + Bindings = new[] + { + new ComputeBinding(0, _workingDepth, ComputeAccess.Sampled, 0, DepthLevels), + new ComputeBinding(1, depthTexture, ComputeAccess.Sampled), + new ComputeBinding(2, normalTexture, ComputeAccess.Sampled), + new ComputeBinding(3, _hilbert, ComputeAccess.Sampled), + new ComputeBinding(4, _workingTerm, ComputeAccess.StorageWrite), + new ComputeBinding(5, _edges, ComputeAccess.StorageWrite), + }, + Dispatches = new[] { ComputeDispatch.Covering(4, push) }, + })) return Fail(_device.GetError()); + + uint passes = Math.Clamp(settings.DenoisePasses, 1, 3); + int source = _workingTerm; + for (uint pass = 0; pass < passes; pass++) + { + bool final = pass + 1 == passes; + int destination = final ? _output : pass % 2 == 0 ? _scratchA : _scratchB; + if (!_device.RecordComputePass(new ComputePassDeclaration + { + Name = final ? "gtao-denoise" : "gtao-denoise-pre", + ProgramId = _denoise, + Specialization = GtaoSettings.DenoiseSpecialization(final), + Bindings = new[] + { + new ComputeBinding(0, source, ComputeAccess.Sampled), + new ComputeBinding(1, _edges, ComputeAccess.Sampled), + new ComputeBinding(2, depthTexture, ComputeAccess.Sampled), + new ComputeBinding(3, normalTexture, ComputeAccess.Sampled), + new ComputeBinding(4, destination, ComputeAccess.StorageWrite), + }, + Dispatches = new[] { ComputeDispatch.Covering(4, push) }, + })) return Fail(_device.GetError()); + source = destination; + } + return _output; + } + + private int Fail(string reason) + { + LastError = string.IsNullOrEmpty(reason) ? "compute pass refused" : reason; + return 0; + } + + private unsafe bool EnsureTargets(uint width, uint height) + { + if (_hilbert == 0) + { + float[] table = HilbertLut.Build(); + fixed (float* data = table) + { + _hilbert = _device.CreateTexture2DRaw(HilbertLut.Width, HilbertLut.Width, 0x822E, (IntPtr)data, 4); + } + } + if (_output != 0 && _width == width && _height == height) return true; + + ReleaseTargets(); + _workingDepth = _device.CreateStorageTexture((int)width, (int)height, Format.R32Sfloat, DepthLevels); + _workingTerm = _device.CreateStorageTexture((int)width, (int)height, Format.R8Unorm); + _edges = _device.CreateStorageTexture((int)width, (int)height, Format.R8Unorm); + _scratchA = _device.CreateStorageTexture((int)width, (int)height, Format.R8Unorm); + _scratchB = _device.CreateStorageTexture((int)width, (int)height, Format.R8Unorm); + _output = _device.CreateStorageTexture((int)width, (int)height, Format.R8Unorm); + _width = width; + _height = height; + return _hilbert != 0 && _device.TextureOf(_workingDepth)?.MipLevels == DepthLevels; + } + + private bool EnsurePrograms() + { + Format depthFormat = _device.TextureOf(_workingDepth)!.Format; + Format termFormat = _device.TextureOf(_workingTerm)!.Format; + if (_prefilter != 0 && depthFormat == _programDepthFormat && termFormat == _programTermFormat) return true; + ReleasePrograms(); + + _prefilter = _device.CreateComputeProgram(GtaoShaderSources.Build("prefilter.comp", depthFormat, termFormat), + "gtao-prefilter", Slots(ComputeSlotKind.Sampled, 1, ComputeSlotKind.Storage, DepthLevels), + GtaoSettings.PushConstantBytes); + _main = _device.CreateComputeProgram(GtaoShaderSources.Build("main.comp", depthFormat, termFormat), + "gtao-main", Slots(ComputeSlotKind.Sampled, 4, ComputeSlotKind.Storage, 2), GtaoSettings.PushConstantBytes); + _denoise = _device.CreateComputeProgram(GtaoShaderSources.Build("denoise.comp", depthFormat, termFormat), + "gtao-denoise", Slots(ComputeSlotKind.Sampled, 4, ComputeSlotKind.Storage, 1), GtaoSettings.PushConstantBytes); + _programDepthFormat = depthFormat; + _programTermFormat = termFormat; + if (_prefilter != 0 && _main != 0 && _denoise != 0) return true; + + LastError = "AO compute shaders failed to compile: " + _device.GetError(); + ProgramsFailed = true; + ReleasePrograms(); + return false; + } + + /// Sampled bindings first, then storage bindings, numbered from 0. + private static ComputeSlot[] Slots(ComputeSlotKind first, int firstCount, ComputeSlotKind second, int secondCount) + { + var slots = new ComputeSlot[firstCount + secondCount]; + for (int i = 0; i < slots.Length; i++) slots[i] = new ComputeSlot((uint)i, i < firstCount ? first : second); + return slots; + } + + /// Frees the size-dependent targets (a framebuffer rebuild); the next render recreates them. + public void ReleaseTargets() + { + foreach (int texture in new[] { _workingDepth, _workingTerm, _edges, _scratchA, _scratchB, _output }) + { + if (texture != 0) _device.DeleteTexture(texture); + } + _workingDepth = _workingTerm = _edges = _scratchA = _scratchB = _output = 0; + _width = _height = 0; + } + + private void ReleasePrograms() + { + if (_prefilter != 0) _device.DeleteComputeProgram(_prefilter); + if (_main != 0) _device.DeleteComputeProgram(_main); + if (_denoise != 0) _device.DeleteComputeProgram(_denoise); + _prefilter = _main = _denoise = 0; + } + + public void Dispose() + { + ReleaseTargets(); + ReleasePrograms(); + if (_hilbert != 0) _device.DeleteTexture(_hilbert); + _hilbert = 0; + } +} + +/// +/// The reconstruction constants of the AO passes from a GL projection +/// (docs/vulkan.md#ambient-occlusion, integration plan step 1): view depth from the +/// [0, 1] depth buffer and view XY from the texel's UV, in the working frame x right, +/// y up, z forward (GL view space mirrored in z). The texel rows run bottom-up (GL order; +/// the device never flips), so unlike XeGTAO's D3D constants the Y terms keep their sign. +/// +/// The jitter columns of the projection (P[2][0], P[2][1]) are folded into the offset, +/// so a jittered G-buffer reconstructs exactly rather than within a sub-pixel. +/// +internal readonly record struct GtaoProjection( + float DepthUnpackMul, float DepthUnpackAdd, + float NdcToViewMulX, float NdcToViewMulY, + float NdcToViewAddX, float NdcToViewAddY) +{ + /// + /// From a column-major GL perspective matrix (m[10] = A, m[14] = B, + /// m[11] = -1); null for anything that is not a perspective projection. + /// + public static GtaoProjection? From(float[]? m) + { + if (m == null || m.Length < 16) return null; + if (MathF.Abs(m[11] + 1f) > 1e-4f || MathF.Abs(m[15]) > 1e-4f) return null; + if (m[0] == 0f || m[5] == 0f || m[14] == 0f) return null; + + float a = m[10]; + float b = m[14]; + float tanX = 1f / m[0]; + float tanY = 1f / m[5]; + return new GtaoProjection( + -b / 2f, (1f - a) / 2f, + 2f * tanX, 2f * tanY, + (m[8] - 1f) * tanX, (m[9] - 1f) * tanY); + } + + /// View depth (positive, forward) of a [0, 1] depth value; the shader's gtaoViewDepth. + public float ViewDepth(float screenDepth) => DepthUnpackMul / (DepthUnpackAdd - screenDepth); + + /// The working-frame position of a texel UV (row 0 at the bottom) at a view depth; gtaoViewPosition. + public (float X, float Y, float Z) ViewPosition(float u, float v, float viewDepth) => + ((NdcToViewMulX * u + NdcToViewAddX) * viewDepth, (NdcToViewMulY * v + NdcToViewAddY) * viewDepth, viewDepth); +} + +/// +/// The AO compute shaders, sources/shaders-vk/gtao/*.comp, embedded in the renderer +/// assembly (so deploy and the packagers carry them with the DLL). Expands their +/// #include "x.glsl" lines from the same directory and defines the storage format +/// qualifiers of the formats the device chose right after #version. +/// +internal static class GtaoShaderSources +{ + public const string ResourcePrefix = "shaders-vk/gtao/"; + + private static readonly ConcurrentDictionary Raw = new(StringComparer.Ordinal); + private static readonly Regex IncludeLine = new("^[ \\t]*#include[ \\t]+\"([^\"]+)\"[ \\t]*\\r?$", + RegexOptions.Multiline | RegexOptions.CultureInvariant); + + /// A resource of the gtao directory, as committed. + public static string Read(string fileName) => Raw.GetOrAdd(fileName, name => + { + Assembly assembly = typeof(GtaoShaderSources).Assembly; + using Stream? stream = assembly.GetManifestResourceStream(ResourcePrefix + name); + if (stream == null) throw new FileNotFoundException("embedded AO shader missing: " + ResourcePrefix + name); + using var reader = new StreamReader(stream, Encoding.UTF8); + return reader.ReadToEnd(); + }); + + /// + /// The compilable source of : includes expanded (each file + /// once) and GTAO_DEPTH_FORMAT / GTAO_TERM_FORMAT defined. + /// + public static string Build(string fileName, Format depthFormat, Format termFormat) + { + string source = Expand(Read(fileName), new System.Collections.Generic.HashSet(StringComparer.Ordinal)); + int versionEnd = source.IndexOf('\n', source.IndexOf("#version", StringComparison.Ordinal)) + 1; + string defines = "#define GTAO_DEPTH_FORMAT " + Qualifier(depthFormat) + "\n" + + "#define GTAO_TERM_FORMAT " + Qualifier(termFormat) + "\n"; + return source.Insert(versionEnd, defines); + } + + private static string Expand(string source, System.Collections.Generic.HashSet included) => + IncludeLine.Replace(source, match => + { + string name = match.Groups[1].Value; + return included.Add(name) ? Expand(Read(name), included) : ""; + }); + + /// The storage image format qualifier of a format a storage texture can have. + public static string Qualifier(Format format) => format switch + { + Format.R8Unorm => "r8", + Format.R8G8Unorm => "rg8", + Format.R8G8B8A8Unorm => "rgba8", + Format.R16Sfloat => "r16f", + Format.R32Sfloat => "r32f", + Format.R16G16B16A16Sfloat => "rgba16f", + Format.R32G32B32A32Sfloat => "rgba32f", + _ => throw new ArgumentOutOfRangeException(nameof(format), format, "no storage qualifier for this format"), + }; +} + +/// +/// The 64x64 Hilbert index table the AO noise starts from (docs/vulkan.md#ambient-occlusion +/// C.7): generated at startup, never shipped. Neighbouring texels get neighbouring indices, +/// so the R2 sequence the index drives is low-discrepancy across the 3x3 denoise +/// footprint. XeGTAO's HilbertIndex (vaGTAO.hlsl, MIT, Intel), level 6. +/// +internal static class HilbertLut +{ + public const int Width = 64; + + /// The curve index of texel (, ), 0..4095. + public static uint Index(uint x, uint y) + { + uint index = 0; + for (uint level = Width / 2; level > 0; level /= 2) + { + uint regionX = (x & level) > 0 ? 1u : 0u; + uint regionY = (y & level) > 0 ? 1u : 0u; + index += level * level * ((3u * regionX) ^ regionY); + if (regionY == 0) + { + if (regionX == 1) + { + x = Width - 1 - x; + y = Width - 1 - y; + } + (x, y) = (y, x); + } + } + return index; + } + + /// The table in row order as floats, for an R32F texture: every index is exact in a float. + public static float[] Build() + { + var table = new float[Width * Width]; + for (uint y = 0; y < Width; y++) + for (uint x = 0; x < Width; x++) + table[y * Width + x] = Index(x, y); + return table; + } +} diff --git a/Optimum.Render.Vulkan/AmbientOcclusion/GtaoSettings.cs b/Optimum.Render.Vulkan/AmbientOcclusion/GtaoSettings.cs new file mode 100644 index 00000000..917aa822 --- /dev/null +++ b/Optimum.Render.Vulkan/AmbientOcclusion/GtaoSettings.cs @@ -0,0 +1,269 @@ +using System; +using System.Globalization; + +namespace Optimum.Render.Vulkan.AmbientOcclusion; + +/// The per-slice integration (docs/vulkan.md#ambient-occlusion C.3, C.13). +internal enum GtaoIntegration : uint +{ + /// 32-sector bitmask with cosine-CDF sector boundaries (C.4); the default. + BitmaskCosine = 0, + /// 32-sector bitmask with sectors uniform in angle (Therrien's original). + BitmaskUniform = 1, + /// XeGTAO's analytic horizon integral, unchanged. + Horizon = 2, +} + +/// The thickness model of the bitmask (C.5, C.13). WIDTH is not implemented. +internal enum GtaoThickness : uint +{ + Constant = 0, + /// Grows linearly with view distance. + Distance = 1, + /// Distance-scaled and randomised per sample; the default. + Random = 2, +} + +/// The quality presets of C.2 / C.12: slices x steps per side. +internal enum GtaoPreset +{ + /// 1x2 (4 fetches). + Low, + /// 2x2 (8 fetches); handheld candidate A at render resolution. + Medium, + /// 3x3 (18 fetches); the discrete preset. + High, + /// 9x3 (54 fetches), two denoise passes; screenshots. + Ultra, +} + +/// The compose tone (C.11): the albedo hook for a later multi-bounce term. +internal enum GtaoTone +{ + Linear, + /// GTAO 2016 eq. 10; refused unless an albedo texture is bound. + MultiBounce, +} + +/// Specialization constant ids; sources/shaders-vk/gtao/common.glsl is the source of truth. +internal static class GtaoSpecialization +{ + public const int Integration = 0; + public const int SliceCount = 1; + public const int StepsPerSlice = 2; + public const int Thickness = 3; + public const int ClassChannel = 4; + public const int NoiseCycle = 5; + public const int NormalEdges = 6; + public const int FinalApply = 7; + + /// One more than the highest id: the length of a specialization array. + public const int Count = 8; +} + +/// +/// Everything the AO passes are parameterised by: the preset's counts, the C.13 variants +/// (specialization constants) and the uniforms (push constants). Pure, so the packing +/// and the environment overrides are testable without a device. +/// +internal sealed record GtaoSettings +{ + public const int PushConstantBytes = 80; + + public uint SliceCount { get; init; } = 2; + public uint StepsPerSlice { get; init; } = 2; + public uint DenoisePasses { get; init; } = 1; + public GtaoIntegration Integration { get; init; } = GtaoIntegration.BitmaskCosine; + public GtaoThickness Thickness { get; init; } = GtaoThickness.Random; + public bool ClassChannel { get; init; } = true; + /// 64 (XeGTAO) or 61, coprime with the 8 and 32 jitter phases (C.7). + public uint NoiseCycle { get; init; } = 64; + public bool NormalEdges { get; init; } + public GtaoTone Tone { get; init; } = GtaoTone.Linear; + + /// Effect radius in blocks (C.2: start 0.75, tuned in D). + public float EffectRadius { get; init; } = 0.75f; + public float RadiusMultiplier { get; init; } = 1.457f; + public float FalloffRange { get; init; } = 0.615f; + /// Measurement only (C.11): 1.0 is the physically meaningful value. + public float FinalValuePower { get; init; } = 1.0f; + public float SampleDistributionPower { get; init; } = 2.0f; + public float DepthMipSamplingOffset { get; init; } = 3.30f; + /// Solid surfaces, blocks (C.5: a fence post is 0.125-0.25, a block 1). + public float ThicknessSolid { get; init; } = 0.5f; + /// The thin class, blocks (C.5). + public float ThicknessThin { get; init; } = 0.05f; + /// Thickness growth per block of view distance; a starting value for D. + public float ThicknessDistanceScale { get; init; } = 1f / 64f; + /// Vanilla ssao.fsh's fade: clamp(1.2 - z / 250, 0, 1). + public float FarFadeBias { get; init; } = 1.2f; + public float FarFadeDistance { get; init; } = 250f; + public float DenoiseBlurBeta { get; init; } = 1.2f; + + /// + /// The preset's counts. With a temporal accumulator one denoise pass (XeGTAO v1.21, + /// Bevy); without it two (C.7/C.8); Ultra always two (the screenshot row of C.12). + /// + public static GtaoSettings ForPreset(GtaoPreset preset, bool temporal) + { + (uint slices, uint steps) = preset switch + { + GtaoPreset.Low => (1u, 2u), + GtaoPreset.Medium => (2u, 2u), + GtaoPreset.High => (3u, 3u), + _ => (9u, 3u), + }; + uint passes = preset == GtaoPreset.Ultra || !temporal ? 2u : 1u; + return new GtaoSettings { SliceCount = slices, StepsPerSlice = steps, DenoisePasses = passes }; + } + + /// + /// Keep the AO input stable before the temporal scene resolve. The default + /// Medium choice uses XeGTAO's 18-sample quality and two spatial denoise + /// passes; an explicit Low choice stays available for slower GPUs. + /// The caller holds the AO sampling phase fixed between frames. + /// + public static GtaoSettings ForStableTemporal(GtaoPreset preset) => + ForPreset(preset == GtaoPreset.Medium ? GtaoPreset.High : preset, temporal: false); + + /// The persisted preset name ("low", "medium", "high", "ultra"); anything else is Medium. + public static GtaoPreset ParsePreset(string? name) => (name ?? "").Trim().ToLowerInvariant() switch + { + "low" => GtaoPreset.Low, + "high" => GtaoPreset.High, + "ultra" => GtaoPreset.Ultra, + _ => GtaoPreset.Medium, + }; + + /// + /// The C.13 measurement variants from the environment (never persisted): + /// OPTIMUM_AO_INTEGRATION (bitmask-cos | bitmask-uniform | horizon), + /// OPTIMUM_AO_THICKNESS (const | dist | random), OPTIMUM_AO_CLASS_CHANNEL (0 | 1), + /// OPTIMUM_AO_NOISE_CYCLE (64 | 61), OPTIMUM_AO_DENOISE_PASSES (1 | 2 | 3), + /// OPTIMUM_AO_NORMAL_EDGES (0 | 1), OPTIMUM_AO_TONE (linear | multibounce), + /// OPTIMUM_AO_FINAL_POWER and OPTIMUM_AO_RADIUS. Unrecognised values keep the setting. + /// + public GtaoSettings WithEnvironment(Func variable) + { + GtaoSettings result = this; + switch (variable("OPTIMUM_AO_INTEGRATION")?.Trim().ToLowerInvariant()) + { + case "bitmask-cos": result = result with { Integration = GtaoIntegration.BitmaskCosine }; break; + case "bitmask-uniform": result = result with { Integration = GtaoIntegration.BitmaskUniform }; break; + case "horizon": result = result with { Integration = GtaoIntegration.Horizon }; break; + } + switch (variable("OPTIMUM_AO_THICKNESS")?.Trim().ToLowerInvariant()) + { + case "const": result = result with { Thickness = GtaoThickness.Constant }; break; + case "dist": result = result with { Thickness = GtaoThickness.Distance }; break; + case "random": result = result with { Thickness = GtaoThickness.Random }; break; + } + switch (variable("OPTIMUM_AO_CLASS_CHANNEL")?.Trim()) + { + case "0": result = result with { ClassChannel = false }; break; + case "1": result = result with { ClassChannel = true }; break; + } + switch (variable("OPTIMUM_AO_NOISE_CYCLE")?.Trim()) + { + case "64": result = result with { NoiseCycle = 64 }; break; + case "61": result = result with { NoiseCycle = 61 }; break; + } + switch (variable("OPTIMUM_AO_DENOISE_PASSES")?.Trim()) + { + case "1": result = result with { DenoisePasses = 1 }; break; + case "2": result = result with { DenoisePasses = 2 }; break; + case "3": result = result with { DenoisePasses = 3 }; break; + } + switch (variable("OPTIMUM_AO_NORMAL_EDGES")?.Trim()) + { + case "0": result = result with { NormalEdges = false }; break; + case "1": result = result with { NormalEdges = true }; break; + } + switch (variable("OPTIMUM_AO_TONE")?.Trim().ToLowerInvariant()) + { + case "linear": result = result with { Tone = GtaoTone.Linear }; break; + case "multibounce": result = result with { Tone = GtaoTone.MultiBounce }; break; + } + if (TryParsePositive(variable("OPTIMUM_AO_FINAL_POWER"), out float power)) result = result with { FinalValuePower = power }; + if (TryParsePositive(variable("OPTIMUM_AO_RADIUS"), out float radius)) result = result with { EffectRadius = radius }; + return result; + } + + private static bool TryParsePositive(string? text, out float value) => + float.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out value) && value > 0f && float.IsFinite(value); + + /// + /// The tone the compose pass may use: needs a real + /// albedo (the scene colour here is lit LDR radiance, C.11), so without an albedo + /// texture it is refused and says why. + /// + public GtaoTone EffectiveTone(int albedoTexture, out string? refusal) + { + refusal = null; + if (Tone != GtaoTone.MultiBounce || albedoTexture != 0) return Tone; + refusal = "multi-bounce AO needs an albedo texture; the scene colour is lit radiance, so the linear tone is used"; + return GtaoTone.Linear; + } + + /// The main pass's specialization values, indexed by id. + public uint[] MainSpecialization() + { + var values = new uint[GtaoSpecialization.Count]; + values[GtaoSpecialization.Integration] = (uint)Integration; + values[GtaoSpecialization.SliceCount] = Math.Max(1, SliceCount); + values[GtaoSpecialization.StepsPerSlice] = Math.Max(1, StepsPerSlice); + values[GtaoSpecialization.Thickness] = (uint)Thickness; + values[GtaoSpecialization.ClassChannel] = ClassChannel ? 1u : 0u; + values[GtaoSpecialization.NoiseCycle] = Math.Max(1, NoiseCycle); + values[GtaoSpecialization.NormalEdges] = NormalEdges ? 1u : 0u; + values[GtaoSpecialization.FinalApply] = 1u; + return values; + } + + /// A denoise pass's specialization values: only FINAL_APPLY varies. + public static uint[] DenoiseSpecialization(bool finalApply) + { + var values = new uint[GtaoSpecialization.Count]; + values[GtaoSpecialization.FinalApply] = finalApply ? 1u : 0u; + return values; + } + + /// The 80-byte push block of common.glsl, in declaration order. + public byte[] PushConstants(GtaoProjection projection, uint noiseIndex) + { + var bytes = new byte[PushConstantBytes]; + int offset = 0; + void Float(float value) + { + BitConverter.TryWriteBytes(bytes.AsSpan(offset, 4), value); + offset += 4; + } + void UInt(uint value) + { + BitConverter.TryWriteBytes(bytes.AsSpan(offset, 4), value); + offset += 4; + } + + Float(projection.DepthUnpackMul); + Float(projection.DepthUnpackAdd); + Float(projection.NdcToViewMulX); + Float(projection.NdcToViewMulY); + Float(projection.NdcToViewAddX); + Float(projection.NdcToViewAddY); + Float(EffectRadius); + Float(FalloffRange); + Float(RadiusMultiplier); + Float(FinalValuePower); + Float(SampleDistributionPower); + Float(DepthMipSamplingOffset); + Float(ThicknessSolid); + Float(ThicknessThin); + Float(ThicknessDistanceScale); + Float(FarFadeBias); + Float(1f / FarFadeDistance); + UInt(noiseIndex); + Float(DenoiseBlurBeta); + UInt(0); + return bytes; + } +} diff --git a/Optimum.Render.Vulkan/Core/BindlessTextureTable.cs b/Optimum.Render.Vulkan/Core/BindlessTextureTable.cs new file mode 100644 index 00000000..fc545174 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/BindlessTextureTable.cs @@ -0,0 +1,790 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; +using Optimum.Render.Vulkan.Shaders; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Set 1 of plan decision 9: one descriptor set of combined-image-sampler arrays, +/// one per GLSL sampled type (), every binding +/// PARTIALLY_BOUND | UPDATE_AFTER_BIND, allocated once from its own +/// update-after-bind pool. Shaders index the arrays with slot numbers from push +/// constants. +/// +/// A slot holds one physical texture under one effective sampler state and +/// layout (); decides +/// which. Slot 0 of every array is that kind's placeholder, and every other slot +/// holds the placeholder until it is allocated and again once it is freed, so a +/// stale or out-of-date index samples a defined value instead of undefined memory: +/// opaque black for colour kinds, as OpenGL reads an unbound texture (magenta under +/// poison mode, where an undefined read is meant to be loud), a far-plane depth for +/// shadow kinds, a fixed texel for integer kinds. +/// +/// Writes are queued and applied by in one +/// vkUpdateDescriptorSets: at frame start (, which first +/// writes placeholders back into slots whose retirement completed) and before +/// every submission of the frame, so a slot first resolved while recording is +/// written before the command buffer naming it is submitted. Update-after-bind +/// makes both legal while the set is bound (docs/vulkan.md#bindless-descriptors, +/// sections 2 and 3). Render thread, except . +/// +internal sealed unsafe class BindlessTextureTable : IDisposable +{ + /// + /// Placeholder colour for colour arrays: opaque black, what OpenGL samples from an unbound + /// texture and what the per-program placeholders read before the shared layout. + /// + private static readonly byte[] OpaqueBlack = { 0, 0, 0, 255 }; + + /// The colour placeholder under poison mode, where an undefined read is meant to be loud. + private static readonly byte[] Magenta = { 255, 0, 255, 255 }; + + private readonly record struct PendingWrite(TextureKind Kind, uint Slot, ImageView View, Sampler Sampler, ImageLayout Layout); + + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly BindlessSlotBook _book; + private readonly object _lock = new(); + private readonly int[] _placeholders = new int[BindlessKinds.Count]; + private readonly List _pending = new(); + // (kind, slot) -> index in _pending: a later write to the same slot replaces the earlier one. + private readonly Dictionary<(TextureKind, uint), int> _pendingIndex = new(); + private readonly List<(TextureKind Kind, uint Slot)> _freed = new(); + private DescriptorPool _pool; + private DescriptorSetLayout _layout; + private bool _disposed; + + public DescriptorSetLayout Layout => _layout; + + /// The one set. Bound at . + public DescriptorSet Set { get; } + + /// Lookups that resolved to a placeholder slot. + public long PlaceholderResolutions { get; private set; } + + /// Slot writes applied since creation (the initial placeholder fill excluded). + public long WritesFlushed { get; private set; } + + /// Writes applied by the most recent that had any. + public int LastFlushWrites { get; private set; } + + public BindlessTextureTable(VulkanContext context, TextureManager textures, ITimelineClock clock) + { + _context = context; + _textures = textures; + uint[] capacities = BindlessKinds.ClampCapacities(context.Capabilities.DescriptorIndexing, + DescriptorIndexingFloor.FrameTextures); + _book = new BindlessSlotBook(clock, capacities); + + try + { + _layout = CreateSetLayout(context, capacities); + _pool = CreatePool(capacities); + Set = AllocateSet(); + CreatePlaceholders(); + FillWithPlaceholders(capacities); + } + catch + { + DestroyObjects(); + throw; + } + } + + public uint CapacityOf(TextureKind kind) => _book.CapacityOf(kind); + + public int LiveSlots(TextureKind kind) + { + lock (_lock) return _book.LiveSlots(kind); + } + + public int PendingRetirements + { + get { lock (_lock) return _book.PendingRetirements; } + } + + public int PendingWrites + { + get { lock (_lock) return _pending.Count; } + } + + /// The texture id of a kind's placeholder. Tests and diagnostics. + public int PlaceholderTextureId(TextureKind kind) => _placeholders[(int)kind]; + + /// for what a texture id resolves to now, aliasing included. + public uint Resolve(int textureId, TextureKind kind, SamplerState state, + ImageLayout layout = ImageLayout.ShaderReadOnlyOptimal) => + Resolve(_textures.Get(textureId), kind, state, layout); + + /// + /// The slot a shader samples through as + /// with in . + /// A new slot's write is queued; the caller records draws with the index and the + /// write lands before their submission. 0 (the placeholder) for no texture, a + /// texture that cannot sit behind the kind, or a full array. + /// + public uint Resolve(VulkanTexture? texture, TextureKind kind, SamplerState state, + ImageLayout layout = ImageLayout.ShaderReadOnlyOptimal) + { + if (texture == null || !BindlessKinds.Suits(TextureShape.Of(texture), kind)) + { + NotePlaceholder(); + return 0; + } + + SamplerState effective = BindlessKinds.EffectiveState(state, kind); + lock (_lock) + { + // Read under the lock Release takes: a delete that set the flag after this + // read blocks in Release until the slot below exists, and then retires it. + if (texture.Released) + { + NotePlaceholder(); + return 0; + } + uint slot = _book.Acquire(new BindlessSlotKey(texture.Id, kind, effective, layout), out bool created); + if (slot == 0) + { + NotePlaceholder(); + return 0; + } + if (created) Queue(new PendingWrite(kind, slot, texture.View, _textures.Samplers.Get(effective), layout)); + return slot; + } + } + + private void NotePlaceholder() + { + lock (_lock) PlaceholderResolutions++; + VulkanStats.NoteBindlessPlaceholderResolution(); + } + + /// + /// Retires every slot of a deleted physical texture against the Frame value + /// recorded now. Any thread (). + /// + public void Release(ulong textureId) + { + lock (_lock) _book.Release(textureId); + } + + /// + /// Frame start, after the ring's wait and collection: slots whose retirement + /// completed get their placeholder back and return to the free lists, then + /// every queued write is applied. + /// + public int BeginFrame() + { + lock (_lock) + { + _freed.Clear(); + _book.Collect(_freed); + foreach ((TextureKind kind, uint slot) in _freed) QueuePlaceholder(kind, slot); + return Flush(); + } + } + + /// Applies every queued write in one vkUpdateDescriptorSets. Returns how many. + public int Flush() + { + lock (_lock) + { + int count = _pending.Count; + if (count == 0) return 0; + + var images = new DescriptorImageInfo[count]; + var writes = new WriteDescriptorSet[count]; + fixed (DescriptorImageInfo* imagesPtr = images) + fixed (WriteDescriptorSet* writesPtr = writes) + { + for (int i = 0; i < count; i++) + { + PendingWrite write = _pending[i]; + imagesPtr[i] = new DescriptorImageInfo(write.Sampler, write.View, write.Layout); + writesPtr[i] = new WriteDescriptorSet + { + SType = StructureType.WriteDescriptorSet, + DstSet = Set, + DstBinding = BindlessKinds.BindingOf(write.Kind), + DstArrayElement = write.Slot, + DescriptorCount = 1, + DescriptorType = DescriptorType.CombinedImageSampler, + PImageInfo = imagesPtr + i, + }; + } + _context.Api.UpdateDescriptorSets(_context.Device, (uint)count, writesPtr, 0, null); + } + + _pending.Clear(); + _pendingIndex.Clear(); + WritesFlushed += count; + LastFlushWrites = count; + VulkanStats.NoteBindlessFlush(count); + return count; + } + } + + private void Queue(in PendingWrite write) + { + if (_pendingIndex.TryGetValue((write.Kind, write.Slot), out int index)) + { + _pending[index] = write; + return; + } + _pendingIndex.Add((write.Kind, write.Slot), _pending.Count); + _pending.Add(write); + } + + private void QueuePlaceholder(TextureKind kind, uint slot) + { + (ImageView view, Sampler sampler) = PlaceholderDescriptor(kind); + Queue(new PendingWrite(kind, slot, view, sampler, ImageLayout.ShaderReadOnlyOptimal)); + } + + private (ImageView View, Sampler Sampler) PlaceholderDescriptor(TextureKind kind) + { + VulkanTexture placeholder = _textures.Get(_placeholders[(int)kind]) + ?? throw new InvalidOperationException("bindless placeholder for " + kind + " is gone"); + return (placeholder.View, _textures.Samplers.Get(BindlessKinds.EffectiveState(SamplerState.Default, kind))); + } + + // ---------------------------------------------------------------- creation + + /// Set 1's layout for the given per-kind capacities. The table's own, and a standalone shared layout's. + internal static DescriptorSetLayout CreateSetLayout(VulkanContext context, uint[] capacities) + { + var bindings = new DescriptorSetLayoutBinding[BindlessKinds.Count]; + var flags = new DescriptorBindingFlags[BindlessKinds.Count]; + for (int i = 0; i < bindings.Length; i++) + { + bindings[i] = new DescriptorSetLayoutBinding + { + Binding = BindlessKinds.BindingOf((TextureKind)i), + DescriptorType = DescriptorType.CombinedImageSampler, + DescriptorCount = capacities[i], + StageFlags = SharedPipelineLayout.Stages, + }; + flags[i] = DescriptorBindingFlags.PartiallyBoundBit | DescriptorBindingFlags.UpdateAfterBindBit; + } + + fixed (DescriptorSetLayoutBinding* bindingsPtr = bindings) + fixed (DescriptorBindingFlags* flagsPtr = flags) + { + var bindingFlags = new DescriptorSetLayoutBindingFlagsCreateInfo + { + SType = StructureType.DescriptorSetLayoutBindingFlagsCreateInfo, + BindingCount = (uint)flags.Length, + PBindingFlags = flagsPtr, + }; + var info = new DescriptorSetLayoutCreateInfo + { + SType = StructureType.DescriptorSetLayoutCreateInfo, + PNext = &bindingFlags, + Flags = DescriptorSetLayoutCreateFlags.UpdateAfterBindPoolBit, + BindingCount = (uint)bindings.Length, + PBindings = bindingsPtr, + }; + DescriptorSetLayout layout; + VulkanResult.Check(context.Api.CreateDescriptorSetLayout(context.Device, &info, null, &layout), + "vkCreateDescriptorSetLayout for the bindless texture set"); + return layout; + } + } + + private DescriptorPool CreatePool(uint[] capacities) + { + uint total = 0; + foreach (uint capacity in capacities) total += capacity; + var size = new DescriptorPoolSize(DescriptorType.CombinedImageSampler, total); + var info = new DescriptorPoolCreateInfo + { + SType = StructureType.DescriptorPoolCreateInfo, + Flags = DescriptorPoolCreateFlags.UpdateAfterBindBit, + MaxSets = 1, + PoolSizeCount = 1, + PPoolSizes = &size, + }; + DescriptorPool pool; + VulkanResult.Check(_context.Api.CreateDescriptorPool(_context.Device, &info, null, &pool), + "vkCreateDescriptorPool for the bindless texture set"); + return pool; + } + + private DescriptorSet AllocateSet() + { + DescriptorSetLayout layout = _layout; + var info = new DescriptorSetAllocateInfo + { + SType = StructureType.DescriptorSetAllocateInfo, + DescriptorPool = _pool, + DescriptorSetCount = 1, + PSetLayouts = &layout, + }; + DescriptorSet set; + VulkanResult.Check(_context.Api.AllocateDescriptorSets(_context.Device, &info, &set), + "vkAllocateDescriptorSets for the bindless texture set"); + return set; + } + + /// + /// One-texel textures of each kind's view type and format class. Their pixels + /// ride the upload batch, which runs before the first frame that can sample them. + /// + private void CreatePlaceholders() + { + fixed (byte* magenta = _context.PoisonFreshResources ? Magenta : OpaqueBlack) + { + _placeholders[(int)TextureKind.Texture2D] = Colour(_textures.Create(1, 1, Format.R8G8B8A8Unorm), 1, magenta); + // A single-layer texture gets a 2D view; an array view needs two layers. + _placeholders[(int)TextureKind.Texture2DArray] = + Colour(_textures.Create(1, 1, Format.R8G8B8A8Unorm, layers: 2), 2, magenta); + _placeholders[(int)TextureKind.TextureCube] = + Colour(_textures.Create(1, 1, Format.R8G8B8A8Unorm, layers: 6, cube: true), 6, magenta); + _placeholders[(int)TextureKind.Texture3D] = Colour(_textures.CreateVolume(1, 1, 1, Format.R8G8B8A8Unorm), 1, magenta); + _placeholders[(int)TextureKind.UnsignedTexture2D] = Colour(_textures.Create(1, 1, Format.R8G8B8A8Uint), 1, magenta); + } + + byte[] signed = { 127, 0, 127, 127 }; + fixed (byte* texel = signed) + { + _placeholders[(int)TextureKind.SignedTexture2D] = Colour(_textures.Create(1, 1, Format.R8G8B8A8Sint), 1, texel); + } + + // Depth at the far plane: every comparison against it passes, so nothing is + // shadowed, as with a missing shadow map on GL. + _placeholders[(int)TextureKind.Shadow2D] = Depth(_textures.Create(1, 1, Format.D32Sfloat), 1); + _placeholders[(int)TextureKind.Shadow2DArray] = Depth(_textures.Create(1, 1, Format.D32Sfloat, layers: 2), 2); + _placeholders[(int)TextureKind.ShadowCube] = Depth(_textures.Create(1, 1, Format.D32Sfloat, layers: 6, cube: true), 6); + + for (int i = 0; i < BindlessKinds.Count; i++) + { + VulkanTexture placeholder = _textures.Get(_placeholders[i])!; + if (!BindlessKinds.Suits(TextureShape.Of(placeholder), (TextureKind)i)) + { + throw new InvalidOperationException("bindless placeholder does not suit " + (TextureKind)i); + } + } + } + + private int Colour(int id, uint layers, byte* texel) + { + for (uint layer = 0; layer < layers; layer++) _textures.Upload(id, 0, 0, 0, 1, 1, (IntPtr)texel, 4, layer); + return id; + } + + private int Depth(int id, uint layers) + { + float far = 1f; + for (uint layer = 0; layer < layers; layer++) _textures.Upload(id, 0, 0, 0, 1, 1, (IntPtr)(&far), 4, layer); + VulkanTexture texture = _textures.Get(id)!; + texture.State = texture.State with { CompareEnable = true }; + return id; + } + + /// Writes each binding's placeholder into every element, one descriptor write per binding. + private void FillWithPlaceholders(uint[] capacities) + { + var infos = new DescriptorImageInfo[BindlessKinds.Count][]; + var handles = new System.Runtime.InteropServices.GCHandle[BindlessKinds.Count]; + var writes = new WriteDescriptorSet[BindlessKinds.Count]; + try + { + for (int i = 0; i < BindlessKinds.Count; i++) + { + (ImageView view, Sampler sampler) = PlaceholderDescriptor((TextureKind)i); + infos[i] = new DescriptorImageInfo[capacities[i]]; + Array.Fill(infos[i], new DescriptorImageInfo(sampler, view, ImageLayout.ShaderReadOnlyOptimal)); + handles[i] = System.Runtime.InteropServices.GCHandle.Alloc(infos[i], + System.Runtime.InteropServices.GCHandleType.Pinned); + writes[i] = new WriteDescriptorSet + { + SType = StructureType.WriteDescriptorSet, + DstSet = Set, + DstBinding = BindlessKinds.BindingOf((TextureKind)i), + DstArrayElement = 0, + DescriptorCount = capacities[i], + DescriptorType = DescriptorType.CombinedImageSampler, + PImageInfo = (DescriptorImageInfo*)handles[i].AddrOfPinnedObject(), + }; + } + fixed (WriteDescriptorSet* writesPtr = writes) + { + _context.Api.UpdateDescriptorSets(_context.Device, (uint)writes.Length, writesPtr, 0, null); + } + } + finally + { + foreach (System.Runtime.InteropServices.GCHandle handle in handles) + { + if (handle.IsAllocated) handle.Free(); + } + } + } + + private void DestroyObjects() + { + Vk api = _context.Api; + // Destroying the pool frees the set. The placeholders belong to the texture manager. + if (_pool.Handle != 0) api.DestroyDescriptorPool(_context.Device, _pool, null); + if (_layout.Handle != 0) api.DestroyDescriptorSetLayout(_context.Device, _layout, null); + _pool = default; + _layout = default; + } + + /// Teardown, after the device-idle wait and after every pipeline layout naming . + public void Dispose() + { + if (_disposed) return; + _disposed = true; + DestroyObjects(); + } +} + +/// +/// The GLSL sampled type a bindless slot serves: one set-1 array per kind, in the +/// order of (the enum value is the index +/// into that table, not necessarily the binding number; see ). +/// +internal enum TextureKind +{ + Texture2D = 0, + Texture2DArray = 1, + TextureCube = 2, + Texture3D = 3, + UnsignedTexture2D = 4, + SignedTexture2D = 5, + Shadow2D = 6, + Shadow2DArray = 7, + ShadowCube = 8, +} + +/// What of a texture decides which kinds it may sit behind. +internal readonly record struct TextureShape(Format Format, uint Layers, bool Cube, bool Volume) +{ + public static TextureShape Of(VulkanTexture texture) => + new(texture.Format, texture.Layers, texture.Cube, texture.Volume); +} + +/// +/// The rules that keep a descriptor legal for the array it is written into. +/// The validation layers check none of them for a partially bound array: a view +/// type that does not match the declaration, a Dref sample through a sampler with +/// compareEnable off (or the reverse), or an integer format behind a float sampler +/// all give undefined or poison texels with no message (docs/vulkan.md#bindless-descriptors, +/// section 1, "Hard rules"). They are enforced here, when a slot is created. +/// +internal static class BindlessKinds +{ + public const int Count = 9; + + private static readonly Dictionary ByGlslType = BuildGlslTypes(); + + private static Dictionary BuildGlslTypes() + { + var types = new Dictionary(StringComparer.Ordinal); + for (int i = 0; i < SetConvention.TextureArrays.Length; i++) + { + types.Add(SetConvention.TextureArrays[i].GlslType, (TextureKind)i); + } + return types; + } + + /// The kind whose array a GLSL sampler of reads; false for types set 1 has no array for. + public static bool TryFromGlslType(string glslType, out TextureKind kind) => ByGlslType.TryGetValue(glslType, out kind); + + /// The set-1 binding number of the kind's array. + public static uint BindingOf(TextureKind kind) => (uint)SetConvention.TextureArrays[(int)kind].Value; + + /// The convention's starting size of the kind's array. + public static uint ConventionCapacity(TextureKind kind) => SetConvention.TextureArrays[(int)kind].Capacity; + + public static bool IsShadow(TextureKind kind) => + kind is TextureKind.Shadow2D or TextureKind.Shadow2DArray or TextureKind.ShadowCube; + + public static bool IsInteger(TextureKind kind) => + kind is TextureKind.UnsignedTexture2D or TextureKind.SignedTexture2D; + + /// + /// Whether a texture of can legally sit behind + /// . The dimensionality rules are the ones + /// the draw path applies to every sampler (a GL texture target + /// cannot change): 2D kinds need one layer, array kinds more than one, cube + /// kinds a cube, 3D a volume. Shadow kinds need a depth format, integer kinds + /// the matching signedness, and float kinds a non-integer format (depth reads + /// through sampler2D as it does on GL). + /// + public static bool Suits(TextureShape shape, TextureKind kind) + { + bool plain = !shape.Cube && !shape.Volume; + bool dimensions = kind switch + { + TextureKind.Texture2D or TextureKind.UnsignedTexture2D or TextureKind.SignedTexture2D + or TextureKind.Shadow2D => plain && shape.Layers == 1, + TextureKind.Texture2DArray or TextureKind.Shadow2DArray => plain && shape.Layers > 1, + TextureKind.TextureCube or TextureKind.ShadowCube => shape.Cube, + TextureKind.Texture3D => shape.Volume, + _ => false, + }; + if (!dimensions) return false; + + string name = shape.Format.ToString(); + bool unsigned = name.Contains("Uint", StringComparison.Ordinal); + bool signed = name.Contains("Sint", StringComparison.Ordinal); + return kind switch + { + _ when IsShadow(kind) => TextureManager.IsDepthFormat(shape.Format), + TextureKind.UnsignedTexture2D => unsigned, + TextureKind.SignedTexture2D => signed, + _ => !unsigned && !signed, + }; + } + + /// + /// The sampler state a slot of is written with. GL keeps + /// compare mode on the texture and leaves a mismatch with the sampler declaration + /// undefined; Vulkan turns it into a poison texel the layers never report. The + /// declaration is what the shader samples through, so it decides: shadow kinds + /// compare, the others do not. Integer formats cannot be filtered linearly or + /// anisotropically, so integer kinds sample nearest. + /// + public static SamplerState EffectiveState(SamplerState state, TextureKind kind) + { + state = state with { CompareEnable = IsShadow(kind) }; + if (IsInteger(kind)) + { + state = state with + { + MagFilter = Filter.Nearest, + MinFilter = Filter.Nearest, + MipmapMode = SamplerMipmapMode.Nearest, + MaxAnisotropy = 1f, + }; + } + return state; + } + + /// + /// Per-kind array sizes: the convention's starting sizes when the device's + /// update-after-bind limits hold them all plus + /// (set 0's textures, which count against the same per-stage limits); otherwise + /// each size scaled down in proportion, never below two (the placeholder and one + /// slot). Devices meeting keep the starting sizes. + /// + public static uint[] ClampCapacities(in DescriptorIndexingSupport support, uint reservedSampledImages) + { + ulong limit = Math.Min(Math.Min(support.MaxPerStageDescriptorUpdateAfterBindSampledImages, + support.MaxPerStageDescriptorUpdateAfterBindSamplers), + Math.Min(support.MaxDescriptorSetUpdateAfterBindSampledImages, + support.MaxDescriptorSetUpdateAfterBindSamplers)); + ulong budget = limit > reservedSampledImages ? limit - reservedSampledImages : 0; + + var capacities = new uint[Count]; + ulong total = 0; + for (int i = 0; i < Count; i++) + { + capacities[i] = ConventionCapacity((TextureKind)i); + total += capacities[i]; + } + if (total <= budget) return capacities; + + for (int i = 0; i < Count; i++) + { + capacities[i] = (uint)Math.Max(2UL, capacities[i] * budget / total); + } + return capacities; + } +} + +/// +/// The slots of one set-1 array: a LIFO free list over 1..Capacity-1. Slot 0 +/// holds the array's placeholder and is never handed out, so 0 always means +/// "nothing to sample" to a shader. +/// +internal sealed class BindlessSlotAllocator +{ + private readonly Stack _free = new(); + private uint _next = 1; + + public BindlessSlotAllocator(uint capacity) + { + if (capacity < 1) throw new ArgumentOutOfRangeException(nameof(capacity)); + Capacity = capacity; + } + + public uint Capacity { get; } + + /// Slots handed out and not freed. + public int Live { get; private set; } + + /// The most recently freed slot, else the lowest never used; false when the array is full. + public bool TryAllocate(out uint slot) + { + if (_free.Count > 0) + { + slot = _free.Pop(); + } + else if (_next < Capacity) + { + slot = _next++; + } + else + { + slot = 0; + return false; + } + Live++; + return true; + } + + public void Free(uint slot) + { + if (slot == 0 || slot >= _next) throw new ArgumentOutOfRangeException(nameof(slot), slot, "not an allocated slot"); + _free.Push(slot); + Live--; + } +} + +/// +/// What a slot holds: a combined image sampler of one physical texture +/// (, never reused, so a GL id handed to a new texture +/// or aliased to a transient image cannot reach an old slot), under one effective +/// sampler state and image layout, in one kind's array. +/// +internal readonly record struct BindlessSlotKey(ulong TextureId, TextureKind Kind, SamplerState State, ImageLayout Layout); + +/// +/// The bookkeeping of the bindless table, without a device: which key owns which +/// slot, and when a retired slot may be handed out again. +/// +/// A live slot is never rewritten in place: a draw recorded against it may still +/// be executing. A texture's slot retires when the texture is deleted, or when the +/// texture has more than keys (a changed +/// glTexParameter or a unit-level sampler override each make a new key; the least +/// recently used one retires, so a texture alternating between two states does not +/// churn slots every draw). A retired slot keeps its descriptor until the Frame +/// timeline value recorded at retirement has completed; then the caller writes the +/// placeholder into it and it returns to the free list. +/// +internal sealed class BindlessSlotBook +{ + public const int MaxVariantsPerTexture = 4; + + private readonly record struct Retired(TextureKind Kind, uint Slot, ulong Frame); + + private readonly ITimelineClock _clock; + private readonly BindlessSlotAllocator[] _allocators; + private readonly Dictionary _slots = new(); + // Each texture's keys, least recently used first. + private readonly Dictionary> _byTexture = new(); + private readonly List _retired = new(); + + public BindlessSlotBook(ITimelineClock clock, IReadOnlyList capacities) + { + if (capacities.Count != BindlessKinds.Count) throw new ArgumentException("one capacity per kind", nameof(capacities)); + _clock = clock; + _allocators = new BindlessSlotAllocator[BindlessKinds.Count]; + for (int i = 0; i < _allocators.Length; i++) _allocators[i] = new BindlessSlotAllocator(capacities[i]); + } + + public uint CapacityOf(TextureKind kind) => _allocators[(int)kind].Capacity; + + public int LiveSlots(TextureKind kind) => _allocators[(int)kind].Live; + + /// Retired slots whose Frame value has not completed yet. + public int PendingRetirements => _retired.Count; + + /// Acquisitions that found the kind's array full and got the placeholder. + public long Exhausted { get; private set; } + + /// + /// The slot holding , allocating one when there is none + /// (: the caller writes the descriptor). 0 when the + /// array is full. + /// + public uint Acquire(in BindlessSlotKey key, out bool created) + { + if (_slots.TryGetValue(key, out uint existing)) + { + List keys = _byTexture[key.TextureId]; + int index = keys.IndexOf(key); + if (index != keys.Count - 1) + { + keys.RemoveAt(index); + keys.Add(key); + } + created = false; + return existing; + } + + created = false; + if (!_allocators[(int)key.Kind].TryAllocate(out uint slot)) + { + Exhausted++; + return 0; + } + + _slots.Add(key, slot); + if (!_byTexture.TryGetValue(key.TextureId, out List? variants)) + { + variants = new List(2); + _byTexture.Add(key.TextureId, variants); + } + variants.Add(key); + if (variants.Count > MaxVariantsPerTexture) + { + BindlessSlotKey oldest = variants[0]; + variants.RemoveAt(0); + Retire(oldest); + } + + created = true; + return slot; + } + + /// Retires every slot of a deleted texture. Returns how many. + public int Release(ulong textureId) + { + if (!_byTexture.Remove(textureId, out List? keys)) return 0; + foreach (BindlessSlotKey key in keys) Retire(key); + return keys.Count; + } + + private void Retire(in BindlessSlotKey key) + { + uint slot = _slots[key]; + _slots.Remove(key); + // Every command that could still name the slot carries this value or an older one. + _retired.Add(new Retired(key.Kind, slot, _clock.FrameRecorded)); + } + + /// + /// Frees every retired slot whose Frame value has completed, appending it to + /// so the caller writes the placeholder back before the + /// slot can be allocated and written again. Returns how many. + /// + public int Collect(List<(TextureKind Kind, uint Slot)> freed) + { + if (_retired.Count == 0) return 0; + + ulong completed = _clock.FrameCompleted; + int kept = 0; + int count = 0; + for (int i = 0; i < _retired.Count; i++) + { + Retired entry = _retired[i]; + if (entry.Frame <= completed) + { + _allocators[(int)entry.Kind].Free(entry.Slot); + freed.Add((entry.Kind, entry.Slot)); + count++; + } + else + { + _retired[kept++] = entry; + } + } + _retired.RemoveRange(kept, _retired.Count - kept); + return count; + } +} diff --git a/Optimum.Render.Vulkan/Core/ComputePipelineCache.cs b/Optimum.Render.Vulkan/Core/ComputePipelineCache.cs new file mode 100644 index 00000000..e01bb4ea --- /dev/null +++ b/Optimum.Render.Vulkan/Core/ComputePipelineCache.cs @@ -0,0 +1,392 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Core.Native; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// Whether a compute program's binding is a storage image or a combined image sampler. +internal enum ComputeSlotKind +{ + Sampled, + Storage, +} + +/// One binding of a compute program's pass set. +internal readonly record struct ComputeSlot(uint Binding, ComputeSlotKind Kind); + +/// What a compute program is built from. +internal sealed class ComputeProgramDescription +{ + public string Name = "compute"; + + /// The compiled module (), entry point main. + public byte[] Spirv = Array.Empty(); + + /// The pass set's bindings: set , in any order. + public ComputeSlot[] Slots = Array.Empty(); + + /// Bytes of the compute stage's push constant block; 0 for none. At most 128 (the spec minimum). + public uint PushConstantBytes; + + /// + /// The work group size used to size group counts from an image when the module does not + /// state a literal one (local_size_x_id). A literal local_size_x / + /// local_size_y in the module always wins (), so the + /// two can never disagree. + /// + public uint LocalSizeX = 8; + public uint LocalSizeY = 8; +} + +/// +/// Reads a compute module's literal work group size (OpExecutionMode LocalSize) from its +/// SPIR-V. glslang also emits gl_WorkGroupSize as a composite decorated BuiltIn +/// WorkgroupSize: a plain OpConstantComposite for literal sizes, an OpSpecConstantComposite +/// when the size comes from specialization constants (local_size_x_id). The latter +/// overrides LocalSize (which then reads 1x1x1), so such a module has no literal size and +/// the reader says so. +/// +internal static class SpirvLocalSize +{ + private const uint MagicNumber = 0x07230203; + private const ushort OpExecutionMode = 16; + private const ushort OpSpecConstantComposite = 51; + private const ushort OpDecorate = 71; + private const uint ExecutionModeLocalSize = 17; + private const uint DecorationBuiltIn = 11; + private const uint BuiltInWorkgroupSize = 25; + + public static bool TryRead(ReadOnlySpan spirv, out uint x, out uint y, out uint z) + { + x = y = z = 0; + if (spirv.Length < 20 || spirv.Length % 4 != 0) return false; + ReadOnlySpan words = System.Runtime.InteropServices.MemoryMarshal.Cast(spirv); + if (words[0] != MagicNumber) return false; + + bool found = false; + uint workgroupSizeId = 0; + var specComposites = new System.Collections.Generic.HashSet(); + int index = 5; + while (index < words.Length) + { + uint word = words[index]; + int count = (int)(word >> 16); + ushort opcode = (ushort)(word & 0xFFFF); + if (count == 0 || index + count > words.Length) return false; + if (opcode == OpExecutionMode && count >= 6 && words[index + 2] == ExecutionModeLocalSize) + { + x = words[index + 3]; + y = words[index + 4]; + z = words[index + 5]; + found = true; + } + else if (opcode == OpDecorate && count >= 4 && words[index + 2] == DecorationBuiltIn && + words[index + 3] == BuiltInWorkgroupSize) + { + workgroupSizeId = words[index + 1]; + } + else if (opcode == OpSpecConstantComposite && count >= 3) + { + specComposites.Add(words[index + 2]); + } + index += count; + } + + if (!found || (workgroupSizeId != 0 && specComposites.Contains(workgroupSizeId))) + { + x = y = z = 0; + return false; + } + return true; + } +} + +/// +/// A compute program: its module, the layout of its one pass set (set 0), its pipeline +/// layout with the compute push constant range, and one pipeline per set of +/// specialization constant values. Destroyed on the timeline after the last frame that +/// could have bound it. +/// +internal sealed unsafe class ComputeProgram : IDisposable +{ + /// The set a compute pass's bindings live in. + public const uint PassSet = 0; + + public const uint MaxPushConstantBytes = 128; + + private readonly VulkanContext _context; + private readonly Dictionary _pipelines = new(); + private bool _disposed; + + public int Id { get; } + public string Name { get; } + public ShaderModule Module { get; } + public DescriptorSetLayout SetLayout { get; } + public PipelineLayout Layout { get; } + public ComputeSlot[] Slots { get; } + public uint PushConstantBytes { get; } + public uint LocalSizeX { get; } + public uint LocalSizeY { get; } + + public int PipelineCount => _pipelines.Count; + + public ComputeProgram(VulkanContext context, int id, ComputeProgramDescription description) + { + if (description.Spirv.Length == 0 || description.Spirv.Length % 4 != 0) + throw new ArgumentException("compute program '" + description.Name + "' has no valid SPIR-V"); + if (description.PushConstantBytes > MaxPushConstantBytes) + throw new ArgumentException("compute program '" + description.Name + "' pushes more than " + MaxPushConstantBytes + " bytes"); + + _context = context; + Id = id; + Name = description.Name; + Slots = (ComputeSlot[])description.Slots.Clone(); + PushConstantBytes = description.PushConstantBytes; + if (SpirvLocalSize.TryRead(description.Spirv, out uint localX, out uint localY, out _)) + { + LocalSizeX = Math.Max(1, localX); + LocalSizeY = Math.Max(1, localY); + } + else + { + LocalSizeX = Math.Max(1, description.LocalSizeX); + LocalSizeY = Math.Max(1, description.LocalSizeY); + } + Vk api = context.Api; + + fixed (byte* code = description.Spirv) + { + var moduleInfo = new ShaderModuleCreateInfo + { + SType = StructureType.ShaderModuleCreateInfo, + CodeSize = (nuint)description.Spirv.Length, + PCode = (uint*)code, + }; + ShaderModule module; + VulkanResult.Check(api.CreateShaderModule(context.Device, &moduleInfo, null, &module), + "vkCreateShaderModule for compute program '" + Name + "'"); + Module = module; + } + + var bindings = new DescriptorSetLayoutBinding[Slots.Length]; + for (int i = 0; i < Slots.Length; i++) + { + bindings[i] = new DescriptorSetLayoutBinding + { + Binding = Slots[i].Binding, + DescriptorType = TypeOf(Slots[i].Kind), + DescriptorCount = 1, + StageFlags = ShaderStageFlags.ComputeBit, + }; + } + fixed (DescriptorSetLayoutBinding* bindingsPtr = bindings) + { + var setInfo = new DescriptorSetLayoutCreateInfo + { + SType = StructureType.DescriptorSetLayoutCreateInfo, + BindingCount = (uint)bindings.Length, + PBindings = bindings.Length == 0 ? null : bindingsPtr, + }; + DescriptorSetLayout setLayout; + Result result = api.CreateDescriptorSetLayout(context.Device, &setInfo, null, &setLayout); + if (result != Result.Success) + { + api.DestroyShaderModule(context.Device, Module, null); + VulkanResult.Check(result, "vkCreateDescriptorSetLayout for compute program '" + Name + "'"); + } + SetLayout = setLayout; + } + + DescriptorSetLayout set = SetLayout; + var push = new PushConstantRange + { + StageFlags = ShaderStageFlags.ComputeBit, + Offset = 0, + Size = PushConstantBytes, + }; + var layoutInfo = new PipelineLayoutCreateInfo + { + SType = StructureType.PipelineLayoutCreateInfo, + SetLayoutCount = 1, + PSetLayouts = &set, + PushConstantRangeCount = PushConstantBytes > 0 ? 1u : 0u, + PPushConstantRanges = PushConstantBytes > 0 ? &push : null, + }; + PipelineLayout layout; + Result layoutResult = api.CreatePipelineLayout(context.Device, &layoutInfo, null, &layout); + if (layoutResult != Result.Success) + { + api.DestroyDescriptorSetLayout(context.Device, SetLayout, null); + api.DestroyShaderModule(context.Device, Module, null); + VulkanResult.Check(layoutResult, "vkCreatePipelineLayout for compute program '" + Name + "'"); + } + Layout = layout; + } + + public static DescriptorType TypeOf(ComputeSlotKind kind) => + kind == ComputeSlotKind.Storage ? DescriptorType.StorageImage : DescriptorType.CombinedImageSampler; + + public bool TryGetSlot(uint binding, out ComputeSlot slot) + { + foreach (ComputeSlot candidate in Slots) + { + if (candidate.Binding != binding) continue; + slot = candidate; + return true; + } + slot = default; + return false; + } + + /// The pipeline for these specialization values, compiled through the driver cache on first use. + internal Pipeline PipelineFor(ReadOnlySpan specialization, PipelineCache driverCache, out bool compiled) + { + var probe = new SpecializationKey(specialization.ToArray()); + if (_pipelines.TryGetValue(probe, out Pipeline existing)) + { + compiled = false; + return existing; + } + + Pipeline pipeline = Compile(specialization, driverCache); + _pipelines[probe] = pipeline; + compiled = true; + return pipeline; + } + + private Pipeline Compile(ReadOnlySpan specialization, PipelineCache driverCache) + { + byte* entry = (byte*)SilkMarshal.StringToPtr("main"); + try + { + var entries = new SpecializationMapEntry[specialization.Length]; + for (int i = 0; i < entries.Length; i++) + { + entries[i] = new SpecializationMapEntry((uint)i, (uint)(i * sizeof(uint)), sizeof(uint)); + } + uint[] values = specialization.ToArray(); + + fixed (SpecializationMapEntry* entriesPtr = entries) + fixed (uint* valuesPtr = values) + { + var info = new SpecializationInfo + { + MapEntryCount = (uint)entries.Length, + PMapEntries = entries.Length == 0 ? null : entriesPtr, + DataSize = (nuint)(values.Length * sizeof(uint)), + PData = values.Length == 0 ? null : valuesPtr, + }; + var createInfo = new ComputePipelineCreateInfo + { + SType = StructureType.ComputePipelineCreateInfo, + Stage = new PipelineShaderStageCreateInfo + { + SType = StructureType.PipelineShaderStageCreateInfo, + Stage = ShaderStageFlags.ComputeBit, + Module = Module, + PName = entry, + PSpecializationInfo = entries.Length == 0 ? null : &info, + }, + Layout = Layout, + }; + Pipeline pipeline; + Result result = _context.Api.CreateComputePipelines(_context.Device, driverCache, 1, &createInfo, null, &pipeline); + if (result != Result.Success) + { + throw new InvalidOperationException("vkCreateComputePipelines failed for '" + Name + "': " + result); + } + return pipeline; + } + } + finally + { + SilkMarshal.Free((nint)entry); + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + Vk api = _context.Api; + foreach (Pipeline pipeline in _pipelines.Values) api.DestroyPipeline(_context.Device, pipeline, null); + _pipelines.Clear(); + api.DestroyPipelineLayout(_context.Device, Layout, null); + api.DestroyDescriptorSetLayout(_context.Device, SetLayout, null); + if (Module.Handle != 0) api.DestroyShaderModule(_context.Device, Module, null); + } + + /// Specialization values compared element by element. + private readonly struct SpecializationKey : IEquatable + { + private readonly uint[] _values; + private readonly int _hash; + + public SpecializationKey(uint[] values) + { + _values = values; + var hash = new HashCode(); + foreach (uint value in values) hash.Add(value); + _hash = hash.ToHashCode(); + } + + public bool Equals(SpecializationKey other) => + _hash == other._hash && _values.AsSpan().SequenceEqual(other._values); + + public override bool Equals(object? obj) => obj is SpecializationKey other && Equals(other); + public override int GetHashCode() => _hash; + } +} + +/// +/// The compute programs of a device and their pipelines. Pipelines compile through the +/// driver's the graphics cache owns, so one cache file +/// warms both kinds on the next launch. Render thread only. +/// +internal sealed class ComputePipelineCache : IDisposable +{ + private readonly VulkanContext _context; + private readonly Func _driverCache; + private readonly Dictionary _programs = new(); + private int _nextId = 1; + private bool _disposed; + + public ComputePipelineCache(VulkanContext context, Func driverCache) + { + _context = context; + _driverCache = driverCache; + } + + public int ProgramCount => _programs.Count; + public long Hits { get; private set; } + public long Misses { get; private set; } + + public int Create(ComputeProgramDescription description) + { + int id = _nextId++; + _programs[id] = new ComputeProgram(_context, id, description); + return id; + } + + public ComputeProgram? Get(int id) => _programs.TryGetValue(id, out ComputeProgram? program) ? program : null; + + /// Removes a program; the caller retires it on the timeline (a submitted frame may still bind it). + public ComputeProgram? Remove(int id) => _programs.Remove(id, out ComputeProgram? program) ? program : null; + + public Pipeline PipelineFor(ComputeProgram program, ReadOnlySpan specialization) + { + Pipeline pipeline = program.PipelineFor(specialization, _driverCache(), out bool compiled); + if (compiled) Misses++; + else Hits++; + return pipeline; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + foreach (ComputeProgram program in _programs.Values) program.Dispose(); + _programs.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Core/DescriptorArena.cs b/Optimum.Render.Vulkan/Core/DescriptorArena.cs new file mode 100644 index 00000000..a6323b93 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/DescriptorArena.cs @@ -0,0 +1,256 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// One frame slot's descriptor sets for short-lived resources, reset wholesale +/// when the slot begins its next frame. +/// +/// is the right home for a set that is bound for +/// hundreds of frames; for a set naming a texture the GUI re-creates every few +/// frames it is pure churn: a write, an index entry, an eviction and a deferred +/// individual free. Here the set lives exactly one frame. Within the frame +/// identical contents share a set, so a text texture drawn twice is written +/// once; at the slot's next frame start (after the Frame timeline says the GPU +/// finished the slot's previous frame) every pool is reset in one call each. +/// +/// No eviction is needed: a resource deleted during the frame is destroyed only +/// after that frame completes, and the sets naming it die at the next reset +/// without ever being bound again. +/// +internal sealed unsafe class DescriptorArena : IDisposable +{ + public const uint SetsPerPool = 256; + + private readonly VulkanContext _context; + private readonly List _pools = new(); + private readonly Dictionary _sets = new(); + private int _poolIndex; + private bool _disposed; + + public DescriptorArena(VulkanContext context) => _context = context; + + /// Distinct sets handed out since the last reset. + public int SetsThisFrame => _sets.Count; + + public int PoolCount => _pools.Count; + + /// Lookups that found a set already written this frame. + public long Hits { get; private set; } + + /// Sets allocated and written, over the arena's life. + public long Allocations { get; private set; } + + public long Resets { get; private set; } + + /// + /// Returns every set to the pools. Only once no submitted command buffer that + /// bound one can still execute: the slot's previous frame has completed. + /// + public void Reset() + { + foreach (DescriptorPool pool in _pools) + { + _context.Api.ResetDescriptorPool(_context.Device, pool, 0); + } + _sets.Clear(); + _poolIndex = 0; + Resets++; + } + + public DescriptorSet Get(DescriptorSetContents contents, DescriptorSetLayout layout) + { + if (_sets.TryGetValue(contents, out DescriptorSet existing)) + { + Hits++; + return existing; + } + + DescriptorSet set = Allocate(layout); + DescriptorCache.Write(_context, set, contents); + _sets[contents] = set; + Allocations++; + return set; + } + + private DescriptorSet Allocate(DescriptorSetLayout layout) + { + // Pools fill in order; a pool that refuses (out of sets, or out of one + // descriptor type) is left for the rest of the frame and the next one tried. + while (true) + { + bool freshPool = _poolIndex == _pools.Count; + if (freshPool) + { + _pools.Add(DescriptorCache.CreatePool(_context, SetsPerPool, 0)); + } + + var allocateInfo = new DescriptorSetAllocateInfo + { + SType = StructureType.DescriptorSetAllocateInfo, + DescriptorPool = _pools[_poolIndex], + DescriptorSetCount = 1, + PSetLayouts = &layout, + }; + + DescriptorSet set; + Result result = _context.Api.AllocateDescriptorSets(_context.Device, &allocateInfo, &set); + if (result == Result.Success) return set; + + if (result != Result.ErrorOutOfPoolMemory && result != Result.ErrorFragmentedPool) + { + throw new InvalidOperationException("vkAllocateDescriptorSets failed in the descriptor arena: " + result); + } + + // A pool created for this very set that still refuses would loop forever. + if (freshPool) + { + throw new InvalidOperationException("a descriptor set does not fit an empty arena pool: " + result); + } + _poolIndex++; + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + _sets.Clear(); + foreach (DescriptorPool pool in _pools) + { + _context.Api.DestroyDescriptorPool(_context.Device, pool, null); + } + _pools.Clear(); + } +} + +/// One image descriptor of a compute pass set. +internal readonly record struct ComputeImageWrite(uint Binding, DescriptorType Type, ImageView View, Sampler Sampler, + ImageLayout Layout); + +/// +/// One frame slot's descriptor sets for compute passes, reset wholesale when the slot +/// begins its next frame (after the Frame timeline says its previous frame finished). +/// +/// A compute pass set names per-level views whose layouts change from pass to pass +/// (storage in one, sampled in the next), so the sets live exactly one frame, like +/// 's; the pools carry the storage image type the +/// draw pools do not. +/// +internal sealed unsafe class ComputeDescriptorArena : IDisposable +{ + public const uint SetsPerPool = 64; + public const uint ImagesPerSet = 8; + + private readonly VulkanContext _context; + private readonly List _pools = new(); + private int _poolIndex; + private bool _disposed; + + public ComputeDescriptorArena(VulkanContext context) => _context = context; + + public int PoolCount => _pools.Count; + public long Allocations { get; private set; } + + public void Reset() + { + foreach (DescriptorPool pool in _pools) _context.Api.ResetDescriptorPool(_context.Device, pool, 0); + _poolIndex = 0; + } + + /// Allocates a set of and writes into it. + public DescriptorSet Get(DescriptorSetLayout layout, ReadOnlySpan writes) + { + DescriptorSet set = Allocate(layout); + Allocations++; + if (writes.Length == 0) return set; + + var images = new DescriptorImageInfo[writes.Length]; + var descriptorWrites = new WriteDescriptorSet[writes.Length]; + fixed (DescriptorImageInfo* imagesPtr = images) + fixed (WriteDescriptorSet* writesPtr = descriptorWrites) + { + for (int i = 0; i < writes.Length; i++) + { + ComputeImageWrite write = writes[i]; + images[i] = new DescriptorImageInfo + { + ImageView = write.View, + Sampler = write.Type == DescriptorType.CombinedImageSampler ? write.Sampler : default, + ImageLayout = write.Layout, + }; + descriptorWrites[i] = new WriteDescriptorSet + { + SType = StructureType.WriteDescriptorSet, + DstSet = set, + DstBinding = write.Binding, + DescriptorCount = 1, + DescriptorType = write.Type, + PImageInfo = imagesPtr + i, + }; + } + _context.Api.UpdateDescriptorSets(_context.Device, (uint)writes.Length, writesPtr, 0, null); + } + return set; + } + + private DescriptorSet Allocate(DescriptorSetLayout layout) + { + while (true) + { + bool freshPool = _poolIndex == _pools.Count; + if (freshPool) _pools.Add(CreatePool()); + + var allocateInfo = new DescriptorSetAllocateInfo + { + SType = StructureType.DescriptorSetAllocateInfo, + DescriptorPool = _pools[_poolIndex], + DescriptorSetCount = 1, + PSetLayouts = &layout, + }; + DescriptorSet set; + Result result = _context.Api.AllocateDescriptorSets(_context.Device, &allocateInfo, &set); + if (result == Result.Success) return set; + if (result != Result.ErrorOutOfPoolMemory && result != Result.ErrorFragmentedPool) + { + throw new InvalidOperationException("vkAllocateDescriptorSets failed in the compute arena: " + result); + } + if (freshPool) + { + throw new InvalidOperationException("a compute descriptor set does not fit an empty arena pool: " + result); + } + _poolIndex++; + } + } + + private DescriptorPool CreatePool() + { + var sizes = stackalloc DescriptorPoolSize[2] + { + new DescriptorPoolSize(DescriptorType.StorageImage, SetsPerPool * ImagesPerSet), + new DescriptorPoolSize(DescriptorType.CombinedImageSampler, SetsPerPool * ImagesPerSet), + }; + var createInfo = new DescriptorPoolCreateInfo + { + SType = StructureType.DescriptorPoolCreateInfo, + PoolSizeCount = 2, + PPoolSizes = sizes, + MaxSets = SetsPerPool, + }; + DescriptorPool pool; + VulkanResult.Check(_context.Api.CreateDescriptorPool(_context.Device, &createInfo, null, &pool), + "vkCreateDescriptorPool for the compute arena"); + return pool; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + foreach (DescriptorPool pool in _pools) _context.Api.DestroyDescriptorPool(_context.Device, pool, null); + _pools.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Core/DescriptorCache.cs b/Optimum.Render.Vulkan/Core/DescriptorCache.cs new file mode 100644 index 00000000..411be914 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/DescriptorCache.cs @@ -0,0 +1,450 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// One combined image sampler binding. +/// +/// is the texture's lifetime id, which is what +/// separates this binding from a later texture that inherits the same view +/// handle. Zero means the resource is permanent and needs no tracking. +/// +internal readonly record struct SamplerBindingValue( + uint Binding, ImageView View, Sampler Sampler, ulong Resource = 0, + ImageLayout Layout = ImageLayout.ShaderReadOnlyOptimal); + +/// One buffer binding. as for samplers. +internal readonly record struct BufferBindingValue( + uint Binding, Buffer Buffer, ulong Offset, ulong Range, ulong Resource = 0); + +/// +/// The contents of one descriptor set, used as a cache key. +/// +/// A set is written once and never updated, so identical contents can always +/// share a handle. That immutability is what makes the cache safe: there is no +/// moment where a set the GPU is reading gets rewritten. +/// +internal sealed class DescriptorSetContents : IEquatable +{ + public int ProgramId { get; } + public int SetIndex { get; } + public SamplerBindingValue[] Samplers { get; } + public BufferBindingValue[] Buffers { get; } + + private readonly int _hash; + + public DescriptorSetContents( + int programId, int setIndex, SamplerBindingValue[] samplers, BufferBindingValue[] buffers) + { + ProgramId = programId; + SetIndex = setIndex; + Samplers = samplers; + Buffers = buffers; + + var hash = new HashCode(); + hash.Add(programId); + hash.Add(setIndex); + foreach (SamplerBindingValue sampler in samplers) + { + hash.Add(sampler.Binding); + hash.Add(sampler.View.Handle); + hash.Add(sampler.Sampler.Handle); + hash.Add(sampler.Resource); + hash.Add((int)sampler.Layout); + } + foreach (BufferBindingValue buffer in buffers) + { + hash.Add(buffer.Binding); + hash.Add(buffer.Buffer.Handle); + hash.Add(buffer.Offset); + hash.Add(buffer.Range); + hash.Add(buffer.Resource); + } + _hash = hash.ToHashCode(); + } + + public bool Equals(DescriptorSetContents? other) + { + if (other is null || other._hash != _hash) return false; + if (ProgramId != other.ProgramId || SetIndex != other.SetIndex) return false; + if (Samplers.Length != other.Samplers.Length) return false; + if (Buffers.Length != other.Buffers.Length) return false; + + for (int i = 0; i < Samplers.Length; i++) + { + if (!Samplers[i].Equals(other.Samplers[i])) return false; + } + for (int i = 0; i < Buffers.Length; i++) + { + if (!Buffers[i].Equals(other.Buffers[i])) return false; + } + return true; + } + + public override bool Equals(object? obj) => Equals(obj as DescriptorSetContents); + public override int GetHashCode() => _hash; +} + +/// +/// Hands out descriptor sets, reusing them whenever the same bindings come back. +/// +/// This is the single largest CPU win available in a Vulkan backend for a game +/// like this one. Chunk rendering binds the same terrain atlas thousands of times +/// per frame; without a cache each of those is a set allocation and a write, and +/// with one they are a dictionary lookup. Published measurements put descriptor +/// caching at roughly a third off frame time in CPU-heavy scenes. +/// +/// Sets are immutable once written, so a set can outlive any number of frames +/// safely - as long as the resources it names do. A set that names a deleted +/// texture is the one thing this cache must never serve again: the driver may +/// give the next texture the same view handle, and a lookup by handle would then +/// hand a draw a set pointing at freed memory. That is why the key carries each +/// resource's lifetime id and why a deleted resource evicts its sets, with the +/// actual free deferred until no frame can still be reading them. +/// +/// The working set is bounded by how many distinct texture combinations the +/// game actually uses at once - a few hundred, not a few hundred thousand - and +/// eviction keeps churn, such as the GUI's re-rendered text, from growing it. +/// +internal sealed unsafe class DescriptorCache : IDisposable +{ + private const uint SetsPerPool = 512; + + /// A pool and how many sets it can still hand out. + private sealed class PoolSlot + { + public DescriptorPool Pool; + public uint Remaining; + } + + private readonly record struct CachedSet(DescriptorSet Set, PoolSlot Pool); + + private readonly VulkanContext _context; + private readonly Dictionary _sets = new(); + private readonly List _pools = new(); + private PoolSlot? _current; + + /// Every cached key that names a given resource, for eviction. + private readonly Dictionary> _byResource = new(); + + /// Resources deleted since the last collection. Any thread may add. + private readonly System.Collections.Concurrent.ConcurrentQueue _pendingReleases = new(); + + private bool _disposed; + + public int Count => _sets.Count; + public long Hits { get; private set; } + public long Misses { get; private set; } + + public DescriptorCache(VulkanContext context) => _context = context; + + public DescriptorSet Get(DescriptorSetContents contents, DescriptorSetLayout layout) + { + if (_sets.TryGetValue(contents, out CachedSet existing)) + { + Hits++; + return existing.Set; + } + + Misses++; + CachedSet cached = Allocate(layout); + Write(_context, cached.Set, contents); + _sets[contents] = cached; + Index(contents); + return cached.Set; + } + + /// + /// Notes that a resource is going away, so no set naming it is handed out + /// again. Safe from any thread; the sets themselves are reclaimed on the + /// render thread by . + /// + public void Release(ulong resource) + { + if (resource != 0) _pendingReleases.Enqueue(resource); + } + + /// + /// Drops every set that names a released resource and returns the work of + /// freeing them, or null when there is none. + /// + /// The sets leave the dictionary here, so no draw recorded from now on can + /// bind them. A frame still executing may be reading one, though, so the + /// caller hands the result to the frame ring and the free itself happens + /// once that frame's fence has signalled. Render thread only. + /// + public IDisposable? CollectReleases() + { + List? doomed = null; + + while (_pendingReleases.TryDequeue(out ulong resource)) + { + if (!_byResource.Remove(resource, out List? keys)) continue; + + foreach (DescriptorSetContents key in keys) + { + // A set naming the resource twice is listed twice, and the + // second removal simply finds nothing. + if (!_sets.Remove(key, out CachedSet cached)) continue; + + Unindex(key, resource); + (doomed ??= new List()).Add(cached); + } + } + + return doomed == null ? null : new FreedSets(this, doomed); + } + + private void Index(DescriptorSetContents contents) + { + foreach (SamplerBindingValue sampler in contents.Samplers) IndexResource(sampler.Resource, contents); + foreach (BufferBindingValue buffer in contents.Buffers) IndexResource(buffer.Resource, contents); + } + + private void IndexResource(ulong resource, DescriptorSetContents contents) + { + if (resource == 0) return; + + if (!_byResource.TryGetValue(resource, out List? keys)) + { + keys = new List(1); + _byResource[resource] = keys; + } + keys.Add(contents); + } + + /// + /// Removes a key from the lists of every other resource it names, so a + /// long-lived resource does not accumulate keys evicted on account of the + /// short-lived ones sampled alongside it. + /// + private void Unindex(DescriptorSetContents key, ulong except) + { + foreach (SamplerBindingValue sampler in key.Samplers) UnindexResource(sampler.Resource, except, key); + foreach (BufferBindingValue buffer in key.Buffers) UnindexResource(buffer.Resource, except, key); + } + + private void UnindexResource(ulong resource, ulong except, DescriptorSetContents key) + { + if (resource == 0 || resource == except) return; + if (!_byResource.TryGetValue(resource, out List? keys)) return; + + keys.Remove(key); + if (keys.Count == 0) _byResource.Remove(resource); + } + + /// Frees a batch of sets back to their pools, once it is safe to. + private sealed class FreedSets : IDisposable + { + private readonly DescriptorCache _cache; + private readonly List _sets; + + public FreedSets(DescriptorCache cache, List sets) + { + _cache = cache; + _sets = sets; + } + + public void Dispose() => _cache.Free(_sets); + } + + private void Free(List sets) + { + // The ring can drain after the cache is gone, and the pools with it. + if (_disposed) return; + + foreach (CachedSet cached in sets) + { + DescriptorSet set = cached.Set; + _context.Api.FreeDescriptorSets(_context.Device, cached.Pool.Pool, 1, &set); + cached.Pool.Remaining++; + } + } + + private CachedSet Allocate(DescriptorSetLayout layout) + { + // Freed sets hand capacity back to whichever pool they came from, so + // any pool with room will do, not only the newest. + PoolSlot? slot = _current is { Remaining: > 0 } ? _current : null; + if (slot == null) + { + foreach (PoolSlot candidate in _pools) + { + if (candidate.Remaining > 0) + { + slot = candidate; + break; + } + } + } + slot ??= GrowPool(); + + Result result = AllocateFrom(slot, layout, out DescriptorSet set); + + // A pool can fail before its nominal capacity when one layout uses more + // of a type than the pool budgeted, or when frees have fragmented it. + // Growing and retrying once is the documented way to handle that; the + // pool that refused is written off rather than asked again every miss. + if (result != Result.Success) + { + slot.Remaining = 0; + slot = GrowPool(); + result = AllocateFrom(slot, layout, out set); + if (result != Result.Success) + { + throw new InvalidOperationException("vkAllocateDescriptorSets failed: " + result); + } + } + + slot.Remaining--; + _current = slot; + return new CachedSet(set, slot); + } + + private Result AllocateFrom(PoolSlot slot, DescriptorSetLayout layout, out DescriptorSet set) + { + var allocateInfo = new DescriptorSetAllocateInfo + { + SType = StructureType.DescriptorSetAllocateInfo, + DescriptorPool = slot.Pool, + DescriptorSetCount = 1, + PSetLayouts = &layout, + }; + + DescriptorSet allocated; + Result result = _context.Api.AllocateDescriptorSets(_context.Device, &allocateInfo, &allocated); + set = allocated; + return result; + } + + private PoolSlot GrowPool() + { + // Evicted sets are freed individually, which a pool has to allow. + DescriptorPool pool = CreatePool(_context, SetsPerPool, DescriptorPoolCreateFlags.FreeDescriptorSetBit); + + var slot = new PoolSlot { Pool = pool, Remaining = SetsPerPool }; + _pools.Add(slot); + _current = slot; + return slot; + } + + /// A pool sized for the shared layout's set 0 and set 2; shared with . + internal static DescriptorPool CreatePool(VulkanContext context, uint maxSets, DescriptorPoolCreateFlags flags) + { + // A pool can only satisfy the descriptor types it was sized for. Without + // a size for a type the allocation fails, the set is never written, and + // the first draw that uses it takes the device down. Set 0 holds one + // dynamic uniform buffer and the frame textures; set 2 one dynamic + // uniform buffer (the record) and every other binding a storage buffer. + var sizes = stackalloc DescriptorPoolSize[3] + { + new DescriptorPoolSize(DescriptorType.UniformBufferDynamic, SetsPerPool * 2), + new DescriptorPoolSize(DescriptorType.CombinedImageSampler, SetsPerPool * 8), + new DescriptorPoolSize(DescriptorType.StorageBuffer, SetsPerPool * (uint)SetConvention.StorageSetBindingCount), + }; + + var createInfo = new DescriptorPoolCreateInfo + { + SType = StructureType.DescriptorPoolCreateInfo, + Flags = flags, + PoolSizeCount = 3, + PPoolSizes = sizes, + MaxSets = maxSets, + }; + + if (context.Api.CreateDescriptorPool(context.Device, &createInfo, null, out DescriptorPool pool) + != Result.Success) + { + throw new InvalidOperationException("vkCreateDescriptorPool failed"); + } + return pool; + } + + internal static void Write(VulkanContext context, DescriptorSet set, DescriptorSetContents contents) + { + int writeCount = contents.Samplers.Length + contents.Buffers.Length; + if (writeCount == 0) return; + + var writes = new WriteDescriptorSet[writeCount]; + var imageInfos = new DescriptorImageInfo[contents.Samplers.Length]; + var bufferInfos = new DescriptorBufferInfo[contents.Buffers.Length]; + + fixed (DescriptorImageInfo* imagePtr = imageInfos) + fixed (DescriptorBufferInfo* bufferPtr = bufferInfos) + { + int index = 0; + + for (int i = 0; i < contents.Samplers.Length; i++) + { + SamplerBindingValue sampler = contents.Samplers[i]; + imageInfos[i] = new DescriptorImageInfo + { + ImageView = sampler.View, + Sampler = sampler.Sampler, + // Normally shader-read-only; a depth attachment sampled by + // the pass that has it bound is read through the read-only + // depth layout instead. + ImageLayout = sampler.Layout, + }; + writes[index++] = new WriteDescriptorSet + { + SType = StructureType.WriteDescriptorSet, + DstSet = set, + DstBinding = sampler.Binding, + DescriptorCount = 1, + DescriptorType = DescriptorType.CombinedImageSampler, + PImageInfo = imagePtr + i, + }; + } + + for (int i = 0; i < contents.Buffers.Length; i++) + { + BufferBindingValue buffer = contents.Buffers[i]; + bufferInfos[i] = new DescriptorBufferInfo + { + Buffer = buffer.Buffer, + Offset = buffer.Offset, + Range = buffer.Range, + }; + + // Set 0's one buffer is the frame block, a dynamic uniform buffer; set 2 + // declares its record dynamic and every other binding a storage buffer. + writes[index++] = new WriteDescriptorSet + { + SType = StructureType.WriteDescriptorSet, + DstSet = set, + DstBinding = buffer.Binding, + DescriptorCount = 1, + DescriptorType = contents.SetIndex == SetConvention.StorageSet + ? SharedPipelineLayout.StorageSetDescriptorType(buffer.Binding) + : DescriptorType.UniformBufferDynamic, + PBufferInfo = bufferPtr + i, + }; + } + + fixed (WriteDescriptorSet* writesPtr = writes) + { + context.Api.UpdateDescriptorSets(context.Device, (uint)writeCount, writesPtr, 0, null); + } + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + _sets.Clear(); + _byResource.Clear(); + foreach (PoolSlot slot in _pools) + { + _context.Api.DestroyDescriptorPool(_context.Device, slot.Pool, null); + } + _pools.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Core/FrameRing.cs b/Optimum.Render.Vulkan/Core/FrameRing.cs new file mode 100644 index 00000000..c675f664 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/FrameRing.cs @@ -0,0 +1,516 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +using Buffer = Silk.NET.Vulkan.Buffer; +using Semaphore = Silk.NET.Vulkan.Semaphore; + +namespace Optimum.Render.Vulkan.Core; + +/// Where a uniform upload landed in the ring buffer. +internal readonly record struct RingAllocation(Buffer Buffer, uint Offset, IntPtr Pointer); + +/// The logical frame shared by all submissions while recording it. +internal sealed class LatencySubmitTag +{ + public ulong FrameId; +} + +/// +/// One frame's worth of transient GPU state. +/// +/// Everything here is reset wholesale rather than freed piecemeal: the command +/// pool, and the bump cursor into this slot's slice of the shared uniform ring. A +/// slot is only reused once the Frame timeline says the GPU has finished the +/// last submission that used it ( waits for that). +/// +/// The slot's frame command buffer is submitted together with the open upload +/// batch (), batch first, in one SubmitInfo that +/// signals both timelines: uploads recorded since the last submission run before +/// the frame that uses them, and nothing waits for them. +/// +/// A frame may be submitted in parts (, for a +/// readback that has to see the frame's work so far). Every command buffer +/// carries its own Frame timeline value, and all of them stay in this slot: the +/// uniform cursor keeps counting, so snapshots taken before a partial submit stay +/// valid after it. +/// +internal sealed unsafe class FrameSlot : IDisposable +{ + private readonly VulkanContext _context; + private readonly FrameTimeline _timeline; + private readonly UploadManager _uploads; + private readonly ulong _alignment; + private readonly ulong _regionStart; + private readonly ulong _regionSize; + private readonly VulkanBuffer _uniformRing; + private readonly LatencySubmitTag _latency; + // Allocated once and recycled: resetting the pool returns every one of them + // to the initial state, where it can be begun again. + private readonly List _commandBuffers = new(); + private int _commandBuffersUsed; + private ulong _cursor; + private bool _disposed; + + /// The slot's position in the ring. + public int Index { get; } + + public CommandPool CommandPool { get; } + public CommandBuffer CommandBuffer { get; private set; } + + /// The Frame timeline value the command buffer being recorded signals when submitted. + public ulong FrameValue { get; private set; } + + /// + /// The value of this slot's newest accepted submission, 0 before the first. + /// The next frame to use the slot waits for it. + /// + public ulong LastSignalledValue { get; private set; } + + /// Partial submissions in the current frame. + public int PartialSubmits { get; private set; } + + public FrameSlot(VulkanContext context, FrameTimeline timeline, UploadManager uploads, VulkanBuffer uniformRing, + ulong regionStart, ulong regionSize, int index = 0, LatencySubmitTag? latency = null) + { + _latency = latency ?? new LatencySubmitTag(); + _context = context; + _timeline = timeline; + _uploads = uploads; + _uniformRing = uniformRing; + _regionStart = regionStart; + _regionSize = regionSize; + _alignment = FrameRing.OffsetAlignment(context.Capabilities); + Index = index; + + var poolInfo = new CommandPoolCreateInfo + { + SType = StructureType.CommandPoolCreateInfo, + QueueFamilyIndex = context.GraphicsQueueFamily, + Flags = CommandPoolCreateFlags.TransientBit, + }; + context.Api.CreateCommandPool(context.Device, &poolInfo, null, out CommandPool commandPool); + CommandPool = commandPool; + } + + /// + /// Recycles the slot for a frame whose first command buffer signals + /// . The caller has already waited for the last + /// submission that used the slot, so resetting the pool is legal. + /// + public void Begin(ulong frameValue) + { + _context.Api.ResetCommandPool(_context.Device, CommandPool, 0); + _cursor = 0; + _commandBuffersUsed = 0; + PartialSubmits = 0; + FrameValue = frameValue; + StartCommandBuffer(); + } + + private static long s_recordingSerials; + + /// + /// Unique across every slot and every begin of : a + /// recycled handle gets a new serial, so state remembered per command buffer + /// () never outlives the recording it describes. + /// + public ulong RecordingSerial { get; private set; } + + private void StartCommandBuffer(bool frameCommands = true) + { + Vk api = _context.Api; + CommandBuffer commandBuffer; + if (_commandBuffersUsed < _commandBuffers.Count) + { + commandBuffer = _commandBuffers[_commandBuffersUsed]; + } + else + { + var allocateInfo = new CommandBufferAllocateInfo + { + SType = StructureType.CommandBufferAllocateInfo, + CommandPool = CommandPool, + Level = CommandBufferLevel.Primary, + CommandBufferCount = 1, + }; + VulkanResult.Check(api.AllocateCommandBuffers(_context.Device, &allocateInfo, &commandBuffer), + "vkAllocateCommandBuffers for a frame slot"); + _commandBuffers.Add(commandBuffer); + } + _commandBuffersUsed++; + RecordingSerial = (ulong)System.Threading.Interlocked.Increment(ref s_recordingSerials); + + var begin = new CommandBufferBeginInfo + { + SType = StructureType.CommandBufferBeginInfo, + Flags = CommandBufferUsageFlags.OneTimeSubmitBit, + }; + VulkanResult.Check(api.BeginCommandBuffer(commandBuffer, &begin), + "vkBeginCommandBuffer for a frame slot"); + CommandBuffer = commandBuffer; + // The present command buffer is not the frame's: an upload recorded while + // it is open goes into the batch, which rides the present submission. + if (frameCommands) _uploads.OnFrameCommandsStarted(commandBuffer); + } + + /// + /// Bump-allocates uniform space in this slot's region of the shared ring. + /// Returns false when the region is exhausted, which the caller reports + /// rather than crashing on. + /// + public bool TryAllocateUniforms(int size, out RingAllocation allocation) + { + ulong aligned = (_cursor + _alignment - 1) / _alignment * _alignment; + if (aligned + (ulong)size > _regionSize) + { + allocation = default; + return false; + } + + ulong absolute = _regionStart + aligned; + allocation = new RingAllocation( + _uniformRing.Handle, (uint)absolute, _uniformRing.Mapped + (int)absolute); + _cursor = aligned + (ulong)size; + return true; + } + + /// + /// Submits what the frame has recorded so far and continues in a new command + /// buffer of this slot, under a newly reserved Frame value. Nothing waits and + /// nothing is reset. The caller closes any open rendering scope first (and + /// must not have an occlusion query open). Returns the value just signalled. + /// + public ulong SubmitPartial() + { + ulong submitted = FrameValue; + Submit(default, default, default, 0); + PartialSubmits++; + FrameValue = _timeline.ReserveFrame(); + StartCommandBuffer(); + return submitted; + } + + /// + /// Closes the frame's command buffer and submits it (Submit A), signalling the + /// Frame timeline to . Nothing waits. Returns the value + /// signalled, which the present submission waits on. + /// + /// Every frame that begins must end here: a reserved Frame value that is never + /// signalled holds back every deferred destruction recorded at or after it. + /// + public ulong EndFrameAndSubmit() + { + ulong submitted = FrameValue; + Submit(default, default, default, 0); + VulkanStats.NoteUniformRingUse(_cursor, _regionSize); + return submitted; + } + + /// + /// After and a successful acquire: starts the + /// present command buffer (Submit B) in this slot under a newly reserved Frame + /// value. + /// + public CommandBuffer BeginPresentCommands() + { + FrameValue = _timeline.ReserveFrame(); + StartCommandBuffer(frameCommands: false); + return CommandBuffer; + } + + /// + /// Submits the present command buffer (Submit B): waits on the Frame timeline + /// at (COLOR_ATTACHMENT_OUTPUT) and on the + /// acquire semaphore at (TRANSFER or + /// COLOR_ATTACHMENT_OUTPUT, never ALL_COMMANDS); signals the binary present + /// semaphore and the Frame timeline. Returns the Frame value signalled. + /// + public ulong SubmitPresent(Semaphore acquireSemaphore, PipelineStageFlags acquireStage, + ulong renderValue, Semaphore presentSemaphore) + { + PresentWaitStages.RequireAcquireStage(acquireStage); + ulong submitted = FrameValue; + Submit(acquireSemaphore, acquireStage, presentSemaphore, renderValue); + return submitted; + } + + /// A binary semaphore to wait on (the acquire semaphore), or none. + /// The stage is waited on at. + /// A binary semaphore to signal (the present semaphore), or none. + /// A Frame timeline value to wait on at COLOR_ATTACHMENT_OUTPUT, or 0. + private void Submit(Semaphore waitSemaphore, PipelineStageFlags waitStage, Semaphore signalSemaphore, + ulong frameWaitValue) + { + Vk api = _context.Api; + CommandBuffer commandBuffer = CommandBuffer; + api.EndCommandBuffer(commandBuffer); + + Semaphore* waits = stackalloc Semaphore[2]; + ulong* waitValues = stackalloc ulong[2]; + PipelineStageFlags* waitStages = stackalloc PipelineStageFlags[2]; + uint waitCount = 0; + if (waitSemaphore.Handle != 0) + { + waits[waitCount] = waitSemaphore; + waitValues[waitCount] = 0; + waitStages[waitCount] = waitStage; + waitCount++; + } + if (frameWaitValue != 0) + { + waits[waitCount] = _timeline.Frame; + waitValues[waitCount] = frameWaitValue; + waitStages[waitCount] = PresentWaitStages.FrameWait; + waitCount++; + } + + // Binary present semaphore first (its value is ignored), then the Frame + // timeline, then the Transfer timeline when an upload batch rides along. + Semaphore* signals = stackalloc Semaphore[3]; + ulong* signalValues = stackalloc ulong[3]; + CommandBuffer* commandBuffers = stackalloc CommandBuffer[2]; + + // The queue is shared with the swapchain's present and between-frames + // upload submissions; see QueueLock. Counted as a wait like any other. + long submitStart = VulkanStats.WaitStart(); + // The upload lock is held from taking the batch to the submit, so no + // upload can land in a batch that is already closed, and Transfer values + // reach the queue in the order they were reserved. + _uploads.EnterSubmit(); + try + { + uint signalCount = 0; + if (signalSemaphore.Handle != 0) + { + signals[signalCount] = signalSemaphore; + signalValues[signalCount] = 0; + signalCount++; + } + signals[signalCount] = _timeline.Frame; + signalValues[signalCount] = FrameValue; + signalCount++; + + uint commandBufferCount = 0; + bool uploads = _uploads.TakeOpenBatchLocked(out CommandBuffer uploadCommands, out ulong transferValue); + if (uploads) + { + // First: it runs before the frame command buffer that samples what it wrote. + commandBuffers[commandBufferCount++] = uploadCommands; + signals[signalCount] = _timeline.Transfer; + signalValues[signalCount] = transferValue; + signalCount++; + } + commandBuffers[commandBufferCount++] = commandBuffer; + + var timelineInfo = new TimelineSemaphoreSubmitInfo + { + SType = StructureType.TimelineSemaphoreSubmitInfo, + WaitSemaphoreValueCount = waitCount, + PWaitSemaphoreValues = waitCount == 0 ? null : waitValues, + SignalSemaphoreValueCount = signalCount, + PSignalSemaphoreValues = signalValues, + }; + + var submit = new SubmitInfo + { + SType = StructureType.SubmitInfo, + PNext = &timelineInfo, + CommandBufferCount = commandBufferCount, + PCommandBuffers = commandBuffers, + WaitSemaphoreCount = waitCount, + PWaitSemaphores = waitCount == 0 ? null : waits, + PWaitDstStageMask = waitCount == 0 ? null : waitStages, + SignalSemaphoreCount = signalCount, + PSignalSemaphores = signals, + }; + + lock (_context.QueueLock) + { + VulkanResult.Check(api.QueueSubmit(_context.GraphicsQueue, 1, &submit, default(Fence)), + "vkQueueSubmit for a frame"); + } + if (uploads) _timeline.NoteTransferSubmitted(transferValue); + _uploads.OnFrameCommandsSubmittedLocked(); + } + finally + { + _uploads.ExitSubmit(); + } + VulkanStats.NoteWait(WaitSite.QueueSubmit, submitStart); + _timeline.NoteFrameSubmitted(FrameValue); + LastSignalledValue = FrameValue; + LastSubmittedFrameId = _latency.FrameId; + } + + /// The frame associated with LastSignalledValue, including partial submits. + public ulong LastSubmittedFrameId { get; private set; } + + public ulong UniformBytesUsed => _cursor; + public ulong UniformCapacity => _regionSize; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + _context.Api.DestroyCommandPool(_context.Device, CommandPool, null); + } +} + +/// +/// Rotates through a small number of frame slots, paced by the Frame timeline. +/// +/// Two in flight is the default: enough to keep the GPU fed while the CPU records +/// the next frame, few enough that input latency stays close to what the OpenGL +/// path had. Frame generation will want a third later. +/// +/// Frames use the slots in turn. Before a frame starts, it waits for the newest +/// Frame value its slot signalled: the end of the frame that last used it (or of +/// that frame's last partial submission). Without partial submissions that is +/// frame n - FramesInFlight. That wait is the only CPU wait the ring makes +/// in steady state. +/// +/// The uniform ring is one buffer for the whole ring rather than one per slot, +/// with each slot bump-allocating inside its own slice. That is what lets +/// descriptor sets be written once and reused forever: the set names the buffer, +/// and the per-draw offset travels as a dynamic offset instead. A buffer per slot +/// would mean rewriting every set every frame, which is the cost this design +/// exists to avoid. +/// +internal sealed class FrameRing : IDisposable +{ + private readonly FrameSlot[] _slots; + private readonly VulkanBuffer _uniformRing; + private readonly FrameTimeline _timeline; + private readonly RetireQueue _retired; + private readonly UploadManager _uploads; + private readonly VulkanAllocator _allocator; + private readonly LatencySubmitTag _latency = new(); + private int _index = -1; + private bool _disposed; + + public FrameRing(VulkanContext context, int framesInFlight = 2, ulong uniformRingSize = 32 * 1024 * 1024, + ulong stagingPerSlot = UploadManager.DefaultStagingPerSlot) + { + _timeline = new FrameTimeline(context); + _retired = new RetireQueue(_timeline); + _uploads = new UploadManager(context, _timeline, _retired, framesInFlight, stagingPerSlot); + _allocator = context.Allocator; + // Per-frame dynamic data: the ReBAR class, falling through to host memory + // (counted and logged) when the cap or the device says no. + // Storage usage too: a rewritten program's named blocks read their per-draw + // snapshot from here as std140 storage buffers (shared layout, set 2). + _uniformRing = new VulkanBuffer(context, uniformRingSize, + BufferUsageFlags.UniformBufferBit | BufferUsageFlags.StorageBufferBit, + MemoryPropertyFlags.DeviceLocalBit | MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, + MemoryPoolClass.ReBar); + + // Each region must start on a uniform-offset boundary, otherwise every + // dynamic offset handed out from slot 1 onwards inherits the misalignment. + ulong alignment = OffsetAlignment(context.Capabilities); + ulong regionSize = uniformRingSize / (ulong)framesInFlight / alignment * alignment; + _slots = new FrameSlot[framesInFlight]; + for (int i = 0; i < framesInFlight; i++) + { + _slots[i] = new FrameSlot(context, _timeline, _uploads, _uniformRing, regionSize * (ulong)i, regionSize, i, + _latency); + } + } + + public int FramesInFlight => _slots.Length; + + /// + /// What every submit of this ring is tagged with (seam S4). The device sets + /// the backend once and the frame id once per frame; the slots read it. + /// + public LatencySubmitTag Latency => _latency; + + /// + /// Every ring offset is a legal uniform and storage buffer offset: both limits are + /// powers of two, so the larger is a multiple of the smaller. + /// + internal static ulong OffsetAlignment(VulkanCapabilities capabilities) => + Math.Max(1UL, Math.Max(capabilities.MinUniformBufferOffsetAlignment, capabilities.MinStorageBufferOffsetAlignment)); + + /// The Frame and Transfer timelines every submission signals. + public FrameTimeline Timeline => _timeline; + + /// The upload batches every submission of this ring carries first. + public UploadManager Uploads => _uploads; + + /// The buffer every uniform descriptor points at. + public Buffer UniformBuffer => _uniformRing.Handle; + + public FrameSlot Current => _index < 0 + ? throw new InvalidOperationException("BeginFrame has not been called yet") + : _slots[_index]; + + /// + /// Starts the next frame: reserves its first Frame value, waits for the last + /// submission of the frame that used its slot before, destroys whatever the + /// timelines say is no longer referenced, and recycles the slot. + /// + public FrameSlot BeginFrame() + { + int index = (_index + 1) % _slots.Length; + FrameSlot slot = _slots[index]; + + ulong frameValue = _timeline.ReserveFrame(); + _timeline.WaitForFrame(slot.LastSignalledValue, WaitSite.FramePacing); + _retired.Collect(); + // After the retirements: blocks they emptied start their empty-frame count. + _allocator.AdvanceFrame(); + + _index = index; + slot.Begin(frameValue); + return slot; + } + + /// + /// Submits the current frame's work so far and keeps recording it in the same + /// slot; see . + /// + public ulong SubmitPartial() => Current.SubmitPartial(); + + /// Ends and submits the current frame (Submit A). Pairs with every BeginFrame. Returns its last Frame value. + public ulong EndFrame() => Current.EndFrameAndSubmit(); + + /// Starts the present command buffer; see . + public CommandBuffer BeginPresentCommands() => Current.BeginPresentCommands(); + + /// Submits the present command buffer (Submit B); see . + public ulong SubmitPresent(Semaphore acquireSemaphore, PipelineStageFlags acquireStage, + ulong renderValue, Semaphore presentSemaphore) => + Current.SubmitPresent(acquireSemaphore, acquireStage, renderValue, presentSemaphore); + + /// + /// Queues a resource for destruction once the GPU is done with it. + /// + /// Safe from any thread. The game's VAO and UBO finalizers call Dispose from + /// the finalizer thread, so this cannot assume it is on the render thread; + /// destruction happens at a later BeginFrame, which is. + /// + /// The resource is keyed on the newest Frame and Transfer values reserved so + /// far (every command that could still name it carries one of them or an + /// older one) and destroyed at the first frame start after both completed. + /// + public void DeferDeletion(IDisposable resource) => _retired.Retire(resource); + + public int PendingDeletionCount => _retired.PendingCount; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + // Callers normally wait for the device to go idle first (VulkanDevice.Dispose + // does); this covers the ones that did not, such as a test unwinding from a + // failed assert. The last submission may still name everything below. + _timeline.WaitForSignalledFramesAtTeardown(); + _uploads.Dispose(); + _retired.DisposeAll(); + foreach (FrameSlot slot in _slots) slot.Dispose(); + _uniformRing.Dispose(); + _timeline.Dispose(); + } +} diff --git a/Optimum.Render.Vulkan/Core/GlEnums.cs b/Optimum.Render.Vulkan/Core/GlEnums.cs new file mode 100644 index 00000000..9e653acf --- /dev/null +++ b/Optimum.Render.Vulkan/Core/GlEnums.cs @@ -0,0 +1,186 @@ +using System; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Translates the raw OpenGL constants the game passes around into Vulkan enums. +/// +/// The client and its mods hand these across the backend seam as plain integers - +/// GlStencilFunc(515, 1, 255), GlDepthFunc, BlendFunc(770, 771) +/// - because that is what they already hold and what the GL path consumed. Keeping +/// them as integers at the boundary is what lets a mod calling +/// IRenderAPI.GlStencilFunc work on this backend without knowing it exists. +/// +internal static class GlEnums +{ + // Blend factors. + private const int Zero = 0; + private const int One = 1; + private const int SrcColor = 0x0300; + private const int OneMinusSrcColor = 0x0301; + private const int SrcAlpha = 0x0302; + private const int OneMinusSrcAlpha = 0x0303; + private const int DstAlpha = 0x0304; + private const int OneMinusDstAlpha = 0x0305; + private const int DstColor = 0x0306; + private const int OneMinusDstColor = 0x0307; + private const int SrcAlphaSaturate = 0x0308; + + public static BlendFactor BlendFactorFrom(int glFactor) => glFactor switch + { + Zero => BlendFactor.Zero, + One => BlendFactor.One, + SrcColor => BlendFactor.SrcColor, + OneMinusSrcColor => BlendFactor.OneMinusSrcColor, + SrcAlpha => BlendFactor.SrcAlpha, + OneMinusSrcAlpha => BlendFactor.OneMinusSrcAlpha, + DstAlpha => BlendFactor.DstAlpha, + OneMinusDstAlpha => BlendFactor.OneMinusDstAlpha, + DstColor => BlendFactor.DstColor, + OneMinusDstColor => BlendFactor.OneMinusDstColor, + SrcAlphaSaturate => BlendFactor.SrcAlphaSaturate, + _ => BlendFactor.One, + }; + + public static BlendOp BlendOpFrom(int glEquation) => glEquation switch + { + 0x8006 => BlendOp.Add, // GL_FUNC_ADD + 0x800A => BlendOp.Subtract, // GL_FUNC_SUBTRACT + 0x800B => BlendOp.ReverseSubtract, // GL_FUNC_REVERSE_SUBTRACT + 0x8007 => BlendOp.Min, // GL_MIN + 0x8008 => BlendOp.Max, // GL_MAX + _ => BlendOp.Add, + }; + + public static CompareOp CompareOpFrom(int glFunc) => glFunc switch + { + 0x0200 => CompareOp.Never, + 0x0201 => CompareOp.Less, + 0x0202 => CompareOp.Equal, + 0x0203 => CompareOp.LessOrEqual, + 0x0204 => CompareOp.Greater, + 0x0205 => CompareOp.NotEqual, + 0x0206 => CompareOp.GreaterOrEqual, + 0x0207 => CompareOp.Always, + _ => CompareOp.Less, + }; + + public static StencilOp StencilOpFrom(int glOp) => glOp switch + { + 0x1E00 => StencilOp.Keep, + 0x0000 => StencilOp.Zero, + 0x1E01 => StencilOp.Replace, + 0x1E02 => StencilOp.IncrementAndClamp, + 0x1E03 => StencilOp.DecrementAndClamp, + 0x150A => StencilOp.Invert, + 0x8507 => StencilOp.IncrementAndWrap, + 0x8508 => StencilOp.DecrementAndWrap, + _ => StencilOp.Keep, + }; + + public static PrimitiveTopology TopologyFrom(EnumDrawMode mode) => mode switch + { + EnumDrawMode.Triangles => PrimitiveTopology.TriangleList, + EnumDrawMode.Lines => PrimitiveTopology.LineList, + EnumDrawMode.LineStrip => PrimitiveTopology.LineStrip, + _ => PrimitiveTopology.TriangleList, + }; + + /// + /// Vulkan can only change topology dynamically within a class, so the class + /// is part of the pipeline key while the exact topology is not. + /// + public static int TopologyClassOf(PrimitiveTopology topology) => topology switch + { + PrimitiveTopology.PointList => 0, + PrimitiveTopology.LineList or PrimitiveTopology.LineStrip => 1, + _ => 2, + }; + + public static Format TextureFormatFrom(EnumTextureInternalFormat format) => format switch + { + EnumTextureInternalFormat.Rgba8 => Format.R8G8B8A8Unorm, + EnumTextureInternalFormat.Rgba16f => Format.R16G16B16A16Sfloat, + EnumTextureInternalFormat.R16f => Format.R16Sfloat, + EnumTextureInternalFormat.DepthComponent32 => Format.D32Sfloat, + _ => Format.R8G8B8A8Unorm, + }; + + /// + /// Formats the vanilla framebuffers use that the public enum does not name. + /// SetupDefaultFrameBuffers passes these as raw GL internal formats. + /// + public static Format TextureFormatFromGl(int glInternalFormat) => glInternalFormat switch + { + 0x8058 => Format.R8G8B8A8Unorm, // GL_RGBA8 + 0x881A => Format.R16G16B16A16Sfloat, // GL_RGBA16F + 0x822D => Format.R16Sfloat, // GL_R16F + // TAA stores linear view depth here. Falling back to RGBA8 clamps every + // distance above 1, making the next resolve reject otherwise valid history. + 0x822E => Format.R32Sfloat, // GL_R32F + 0x8C3A => Format.B10G11R11UfloatPack32,// GL_R11F_G11F_B10F + 0x8814 => Format.R32G32B32A32Sfloat, // GL_RGBA32F + 0x805B => Format.R16G16B16A16Unorm, // GL_RGBA16, the cloud map's tile data + 0x8051 => Format.R8G8B8A8Unorm, // GL_RGB8, promoted: RGB is not a + 0x1907 => Format.R8G8B8A8Unorm, // GL_RGB guaranteed attachment format + 0x8CAC => Format.D32Sfloat, // GL_DEPTH_COMPONENT32F + 0x81A5 => Format.D16Unorm, // GL_DEPTH_COMPONENT16 + // GL_BGRA. The GL bodies use it as a source pixel format against an + // RGBA8 internal format; here it names a BGRA-ordered image, so Cairo + // and GUI uploads land with their channels in the right places. + 0x80E1 => Format.B8G8R8A8Unorm, + _ => Format.R8G8B8A8Unorm, + }; + + public static Filter FilterFrom(int glFilter) => glFilter switch + { + 0x2600 => Filter.Nearest, // GL_NEAREST + 0x2601 => Filter.Linear, // GL_LINEAR + _ => Filter.Nearest, + }; + + /// + /// GL's minification filters fold the mipmap mode into the same constant. + /// + public static (Filter Filter, SamplerMipmapMode MipmapMode) MinFilterFrom(int glFilter) => glFilter switch + { + 0x2600 => (Filter.Nearest, SamplerMipmapMode.Nearest), // GL_NEAREST + 0x2601 => (Filter.Linear, SamplerMipmapMode.Nearest), // GL_LINEAR + 0x2700 => (Filter.Nearest, SamplerMipmapMode.Nearest), // NEAREST_MIPMAP_NEAREST + 0x2701 => (Filter.Linear, SamplerMipmapMode.Nearest), // LINEAR_MIPMAP_NEAREST + 0x2702 => (Filter.Nearest, SamplerMipmapMode.Linear), // NEAREST_MIPMAP_LINEAR + 0x2703 => (Filter.Linear, SamplerMipmapMode.Linear), // LINEAR_MIPMAP_LINEAR + _ => (Filter.Nearest, SamplerMipmapMode.Nearest), + }; + + /// + /// Whether a GL min filter samples the mip chain at all. Only the four + /// MIPMAP forms do; GL_NEAREST and GL_LINEAR read level 0 however many + /// levels the texture owns. + /// + public static bool MinFilterUsesMipmaps(int glFilter) => + glFilter is 0x2700 or 0x2701 or 0x2702 or 0x2703; + + public static SamplerAddressMode AddressModeFrom(int glWrap) => glWrap switch + { + 0x2901 => SamplerAddressMode.Repeat, // GL_REPEAT + 0x812F => SamplerAddressMode.ClampToEdge, // GL_CLAMP_TO_EDGE + 0x812D => SamplerAddressMode.ClampToBorder, // GL_CLAMP_TO_BORDER + 0x8370 => SamplerAddressMode.MirroredRepeat, // GL_MIRRORED_REPEAT + _ => SamplerAddressMode.ClampToEdge, + }; + + // Texture parameter names the client actually sets. + public const int TextureMinFilter = 0x2801; + public const int TextureMagFilter = 0x2800; + public const int TextureWrapS = 0x2802; + public const int TextureWrapT = 0x2803; + public const int TextureCompareMode = 0x884C; + public const int TextureLodBias = 0x8501; + public const int TextureMaxLevel = 0x813D; + public const int TextureBorderColor = 0x1004; + public const int TextureCompareModeNone = 0; + public const int TextureCompareRefToTexture = 0x884E; +} diff --git a/Optimum.Render.Vulkan/Core/GpuCheckpoints.cs b/Optimum.Render.Vulkan/Core/GpuCheckpoints.cs new file mode 100644 index 00000000..4299fb71 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/GpuCheckpoints.cs @@ -0,0 +1,99 @@ +using System; + +namespace Optimum.Render.Vulkan.Core; + +/// What a GPU checkpoint marker stands for. +internal enum CheckpointKind : byte +{ + None = 0, + FrameBegin = 1, + Draw = 2, + DrawMulti = 3, + Fullscreen = 4, + Upload = 5, + Mipmaps = 6, + PresentBlit = 7, +} + +/// +/// Packs what a GPU checkpoint refers to into the pointer-sized marker the +/// driver records, and reads it back after a device loss. +/// +/// VK_NV_device_diagnostic_checkpoints stores the marker value and hands it +/// straight back; nobody dereferences it. So it can carry data rather than point +/// at any, which is what makes a checkpoint per draw affordable: nothing is +/// allocated, and nothing has to be kept alive for a loss that may never come. +/// +/// Layout, high to low: 4 bits of kind, 28 bits of "a", 32 bits of "b". The +/// kind is never zero, so neither is the marker, which keeps "no checkpoint" +/// distinguishable from a real one. +/// +internal static class CheckpointMarker +{ + private const int KindShift = 60; + private const int AShift = 32; + private const ulong AMask = (1UL << 28) - 1; + private const ulong BMask = 0xFFFFFFFFUL; + + public static nint Pack(CheckpointKind kind, uint a, uint b) => + (nint)(long)(((ulong)kind << KindShift) | (((ulong)a & AMask) << AShift) | ((ulong)b & BMask)); + + public static CheckpointKind KindOf(nint marker) => (CheckpointKind)((ulong)(long)marker >> KindShift); + public static uint AOf(nint marker) => (uint)(((ulong)(long)marker >> AShift) & AMask); + public static uint BOf(nint marker) => (uint)((ulong)(long)marker & BMask); + + public static nint FrameBegin(uint frame) => Pack(CheckpointKind.FrameBegin, 0, frame); + + public static nint Draw(CheckpointKind kind, int program, int target, int mesh) => + Pack(kind, ((uint)Math.Clamp(target, 0, 0xFFF) << 16) | ((uint)program & 0xFFFF), (uint)mesh); + + public static nint Upload(int texture, uint width, uint height) => + Pack(CheckpointKind.Upload, (uint)texture, (Math.Min(width, 0xFFFFu) << 16) | Math.Min(height, 0xFFFFu)); + + public static nint Mipmaps(int texture, uint levels) => Pack(CheckpointKind.Mipmaps, (uint)texture, levels); + + public static nint PresentBlit(uint image, uint frame) => Pack(CheckpointKind.PresentBlit, image, frame); + + /// Renders a marker for a person, naming the program where one is known. + public static string Describe(nint marker, Func? programName = null) + { + CheckpointKind kind = KindOf(marker); + uint a = AOf(marker); + uint b = BOf(marker); + + switch (kind) + { + case CheckpointKind.FrameBegin: + return "frame " + b + " begins"; + + case CheckpointKind.Draw: + case CheckpointKind.DrawMulti: + case CheckpointKind.Fullscreen: + { + int program = (int)(a & 0xFFFF); + int target = (int)(a >> 16); + string name = programName?.Invoke(program) is { } known ? " '" + known + "'" : ""; + string what = kind switch + { + CheckpointKind.DrawMulti => "multi-draw", + CheckpointKind.Fullscreen => "fullscreen draw", + _ => "draw", + }; + string mesh = kind == CheckpointKind.Fullscreen ? "" : " mesh " + b; + return what + " with program " + program + name + mesh + " into framebuffer " + target; + } + + case CheckpointKind.Upload: + return "upload of " + (b >> 16) + "x" + (b & 0xFFFF) + " texels into texture " + a; + + case CheckpointKind.Mipmaps: + return "mipmap generation for texture " + a + " (" + b + " levels)"; + + case CheckpointKind.PresentBlit: + return "blit of frame " + b + " into swapchain image " + a; + + default: + return "unknown marker 0x" + ((ulong)(long)marker).ToString("x"); + } + } +} diff --git a/Optimum.Render.Vulkan/Core/MeshManager.cs b/Optimum.Render.Vulkan/Core/MeshManager.cs new file mode 100644 index 00000000..b5692e45 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/MeshManager.cs @@ -0,0 +1,570 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// A mesh: one buffer per attribute, plus indices. +/// +/// The per-attribute layout is not a choice - it is how the game allocates. Its +/// mesh allocator makes a separate GL buffer for positions, normals, UVs, +/// colours and flags, and only the four "custom" parts are interleaved. Matching +/// that exactly is what lets the existing upload paths, including the +/// persistently mapped writes the chunk tesselator does, work unchanged. +/// +internal sealed class VulkanMesh : IDisposable +{ + public VulkanBuffer?[] Buffers { get; } = new VulkanBuffer?[MeshManager.MaxBuffers]; + public VulkanBuffer? Indices { get; set; } + + public int IndexCount { get; set; } + public EnumDrawMode DrawMode { get; set; } = EnumDrawMode.Triangles; + public bool Persistent { get; set; } + public bool Ssbo { get; set; } + + public VertexLayoutDescription Layout { get; set; } = VertexLayoutDescription.Empty; + public int LayoutId { get; set; } = -1; + + /// Which buffers actually feed vertex bindings, in binding order. + public List BindingOrder { get; } = new(); + + public void Dispose() + { + foreach (VulkanBuffer? buffer in Buffers) buffer?.Dispose(); + Array.Clear(Buffers); + Indices?.Dispose(); + Indices = null; + } +} + +/// +/// Owns meshes and hands out integer ids, mirroring the GL VAO the game's +/// MeshRef wraps. +/// +internal sealed unsafe class MeshManager : IDisposable +{ + /// xyz, normals, uv, rgba, flags, then the four custom parts. + public const int MaxBuffers = 9; + + public const int BufferXyz = 0; + public const int BufferNormals = 1; + public const int BufferUv = 2; + public const int BufferRgba = 3; + public const int BufferFlags = 4; + public const int BufferCustomFloat = 5; + public const int BufferCustomShort = 6; + public const int BufferCustomInt = 7; + public const int BufferCustomByte = 8; + + private const int GlUnsignedByte = 0x1401; + private const int GlShort = 0x1402; + private const int GlUnsignedShort = 0x1403; + private const int GlUnsignedInt = 0x1405; + + /// + /// The flags and custom-int attributes are fed to shader inputs declared + /// in int. GL let an unsigned pointer feed a signed input - it + /// reinterprets - but Vulkan requires the attribute format's numeric type to + /// match the shader's exactly, so these are signed here. + /// + private const int GlInt = 0x1404; + + private const int GlFloat = 0x1406; + private const int GlInt2101010Rev = 0x8D9F; + + private readonly VulkanContext _context; + private readonly UploadManager? _uploads; + private readonly Interner _layouts = new(); + private readonly List _meshes = new(); + private readonly Stack _freeIds = new(); + private bool _disposed; + + /// + /// The layout of a pass with no vertex buffers, reserved as id 0. + /// + /// The fullscreen post-processing passes generate their vertices from + /// gl_VertexIndex and bind nothing, but they still need a layout id for the + /// pipeline key. Interning the empty layout first guarantees the id exists + /// even before any mesh has been created. + /// + public const int EmptyLayoutId = 0; + + /// + /// Static meshes on device-local memory, filled through the upload manager's + /// staging instead of a host mapping. On since Phase 1B step 5 moved static + /// meshes off ReBAR; it only takes effect with an upload manager. + /// + internal bool DeviceLocalStaticBuffers { get; set; } = true; + + public MeshManager(VulkanContext context, UploadManager? uploads = null) + { + _context = context; + _uploads = uploads; + _meshes.Add(null); // 0 is never a real mesh + + int emptyId = _layouts.Intern(VertexLayoutDescription.Empty); + if (emptyId != EmptyLayoutId) + { + throw new InvalidOperationException("the empty vertex layout must intern first"); + } + } + + public VulkanMesh? Get(int id) => id > 0 && id < _meshes.Count ? _meshes[id] : null; + + public int Count + { + get + { + int live = 0; + foreach (VulkanMesh? mesh in _meshes) + { + if (mesh != null) live++; + } + return live; + } + } + + /// + /// Creates a mesh with the given per-part byte sizes, matching the shape of + /// the game's AllocateEmptyMesh. A part with size 0 is absent, and absent + /// parts do not consume an attribute location - which is what makes the chunk + /// shaders' location numbering line up without any per-shader knowledge here. + /// + public int CreateEmpty( + int xyzSize, int normalsSize, int uvSize, int rgbaSize, int flagsSize, int indicesSize, + CustomMeshDataPartFloat? customFloats, CustomMeshDataPartShort? customShorts, + CustomMeshDataPartByte? customBytes, CustomMeshDataPartInt? customInts, + EnumDrawMode drawMode, bool staticDraw, bool ssbo, bool signedCustomShorts = false) + { + var mesh = new VulkanMesh + { + DrawMode = drawMode, + Persistent = !staticDraw, + Ssbo = ssbo, + }; + + var builder = new VertexLayoutBuilder(); + + // The order here is the order the GL allocator assigns attribute slots. + // With SSBO vertex fetch the xyz slot holds packed face records rather + // than positions: one 64-byte record per four vertices, so 16 bytes per + // vertex where a position is 12. GL sizes that buffer as xyzSize / 12 * 16 + // and so must this, or the last quarter of every pool is out of range - + // which robust buffer access reads back as zeros, collapsing those faces + // onto the origin and stretching their neighbours across the screen. + int xyzSlotSize = ssbo ? xyzSize / 12 * 16 : xyzSize; + AddDedicated(mesh, builder, BufferXyz, xyzSlotSize, 3, GlFloat, normalized: false, integer: false, ssbo); + + // Normals, uv and flags have no vertex binding on the SSBO path: their + // contents ride in the packed face records instead, and GL's SSBO + // allocator creates neither a buffer nor an attribute pointer for them. + // Adding one here would push rgba off location 0, so the shader's + // rgbaLightIn would read the uv stream - block light taken from atlas + // coordinates, which tints the terrain by texture position. + AddDedicated(mesh, builder, BufferNormals, ssbo ? 0 : normalsSize, 4, GlInt2101010Rev, normalized: true, integer: false, ssbo); + AddDedicated(mesh, builder, BufferUv, ssbo ? 0 : uvSize, 2, GlFloat, normalized: false, integer: false, ssbo); + AddDedicated(mesh, builder, BufferRgba, rgbaSize, 4, GlUnsignedByte, normalized: true, integer: false, ssbo); + AddDedicated(mesh, builder, BufferFlags, ssbo ? 0 : flagsSize, 1, GlInt, normalized: false, integer: true, ssbo); + + AddCustom(mesh, builder, BufferCustomFloat, customFloats?.AllocationSize * 4 ?? 0, + customFloats?.InterleaveSizes, customFloats?.InterleaveOffsets, + customFloats?.InterleaveStride ?? 0, GlFloat, false, false, + customFloats?.Instanced ?? false); + + // AllocateEmptyMesh/AddCustoms uses GL_UNSIGNED_SHORT for float + // inputs, including the packed secondary UVs of topsoil. UploadMesh + // uses GL_SHORT instead (legacy clouds rely on signed offsets). + // Integer inputs use GL_SHORT on both paths. + int shortType = signedCustomShorts || customShorts?.Conversion == DataConversion.Integer + ? GlShort : GlUnsignedShort; + AddCustom(mesh, builder, BufferCustomShort, customShorts?.AllocationSize * 2 ?? 0, + customShorts?.InterleaveSizes, customShorts?.InterleaveOffsets, + customShorts?.InterleaveStride ?? 0, shortType, + customShorts?.Conversion == DataConversion.NormalizedFloat, + customShorts?.Conversion == DataConversion.Integer, + customShorts?.Instanced ?? false); + + (int intBytes, int[]? intSizes, int[]? intOffsets, int intStride) = + PruneCustomInts(customInts, ssbo); + + AddCustom(mesh, builder, BufferCustomInt, intBytes, intSizes, intOffsets, intStride, GlInt, + customInts?.Conversion == DataConversion.NormalizedFloat, + customInts?.Conversion == DataConversion.Integer, + customInts?.Instanced ?? false); + + AddCustom(mesh, builder, BufferCustomByte, customBytes?.AllocationSize ?? 0, + customBytes?.InterleaveSizes, customBytes?.InterleaveOffsets, + customBytes?.InterleaveStride ?? 0, GlUnsignedByte, + customBytes?.Conversion == DataConversion.NormalizedFloat, + customBytes?.Conversion == DataConversion.Integer, + customBytes?.Instanced ?? false); + + if (indicesSize > 0) + { + mesh.Indices = CreateBuffer(indicesSize, BufferUsageFlags.IndexBufferBit, mesh.Persistent); + mesh.IndexCount = indicesSize / sizeof(int); + if (ssbo) FillQuadIndices(mesh.Indices); + } + + mesh.Layout = builder.Build(); + mesh.LayoutId = _layouts.Intern(mesh.Layout); + + return Register(mesh); + } + + /// + /// The custom-int part as the SSBO path sees it. GL prunes it there: the + /// first interleaved member is the colormap data, which the face record now + /// carries, so it is dropped, the rest tighten onto half the stride and the + /// buffer halves with them. A part that only ever had that one member is + /// dropped entirely. Off the SSBO path the part passes through unchanged. + /// + private static (int Bytes, int[]? Sizes, int[]? Offsets, int Stride) PruneCustomInts( + CustomMeshDataPartInt? customInts, bool ssbo) + { + if (customInts == null) return (0, null, null, 0); + + int bytes = customInts.AllocationSize * 4; + int[]? sizes = customInts.InterleaveSizes; + int[]? offsets = customInts.InterleaveOffsets; + int stride = customInts.InterleaveStride; + + if (!ssbo) return (bytes, sizes, offsets, stride); + + if (stride <= 4 || sizes == null || sizes.Length < 2) return (0, null, null, 0); + + // Member k reads what member k - 1 used to, because dropping the first + // one shifts every remaining offset down a slot. + return (bytes / 2, sizes[1..], offsets?[..^1], stride / 2); + } + + private void AddDedicated( + VulkanMesh mesh, VertexLayoutBuilder builder, int slot, int byteSize, + int components, int glType, bool normalized, bool integer, bool ssbo) + { + if (byteSize <= 0) return; + + // The SSBO path reads positions through a storage buffer rather than the + // vertex input, so that buffer needs the extra usage bit. + BufferUsageFlags usage = BufferUsageFlags.VertexBufferBit; + if (ssbo && slot == BufferXyz) usage |= BufferUsageFlags.StorageBufferBit; + + mesh.Buffers[slot] = CreateBuffer(byteSize, usage, mesh.Persistent); + + // With SSBO vertex fetch the position buffer is not a vertex binding. + if (ssbo && slot == BufferXyz) return; + + mesh.BindingOrder.Add(slot); + builder.AddDedicated( + VertexLayoutBuilder.FormatFor(components, glType, normalized, integer), + VertexLayoutBuilder.SizeOf(components, glType)); + } + + private void AddCustom( + VulkanMesh mesh, VertexLayoutBuilder builder, int slot, int byteSize, + int[]? interleaveSizes, int[]? interleaveOffsets, int stride, + int glType, bool normalized, bool integer, bool instanced) + { + // Presence of the part decides whether it takes an attribute location, + // not how much data it currently holds. The GL allocator does the same: + // it adds the attribute pointers whenever the part is non-null, and + // AllocationSize returns Count, which is zero for a part that will be + // filled after allocation. Gating on size here would shift every later + // location and silently misfeed the shader. + if (interleaveSizes == null || interleaveSizes.Length == 0) return; + + // Vulkan rejects a zero-sized buffer, so an empty part still gets a + // minimal allocation to keep the binding valid. + mesh.Buffers[slot] = CreateBuffer( + Math.Max(byteSize, 4), BufferUsageFlags.VertexBufferBit, mesh.Persistent); + mesh.BindingOrder.Add(slot); + + var members = new (Format, uint)[interleaveSizes.Length]; + uint packedStride = 0; + for (int i = 0; i < interleaveSizes.Length; i++) + { + uint offset = interleaveOffsets != null && i < interleaveOffsets.Length + ? (uint)interleaveOffsets[i] + : packedStride; + + members[i] = (VertexLayoutBuilder.FormatFor(interleaveSizes[i], glType, normalized, integer), offset); + packedStride += VertexLayoutBuilder.SizeOf(interleaveSizes[i], glType); + } + + builder.AddInterleaved(members, stride > 0 ? (uint)stride : packedStride, instanced); + } + + private VulkanBuffer CreateBuffer(int byteSize, BufferUsageFlags usage, bool persistent) + { + // Phase 1B step 5: a static mesh lives in device-local memory (a type + // that is not host visible, when the device has one), filled through the + // upload manager's staging. It never takes ReBAR, which holds only + // per-frame dynamic data. + if (!persistent && DeviceLocalStaticBuffers && _uploads != null) + { + return new VulkanBuffer(_context, (ulong)byteSize, usage | BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.DeviceLocalBit, MemoryPoolClass.DeviceBuffers); + } + + // A dynamic mesh is host visible and stays mapped, because the game + // writes straight through the pointer while the GPU may still be + // reading - the same lack of synchronisation GL allowed and the chunk + // tesselator relies on. A static mesh with no upload manager to stage + // through (component tests) is host visible too, still off ReBAR. + return new VulkanBuffer(_context, (ulong)byteSize, usage | BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, MemoryPoolClass.DeviceBuffers); + } + + private int Register(VulkanMesh mesh) + { + if (_freeIds.Count > 0) + { + int reused = _freeIds.Pop(); + _meshes[reused] = mesh; + return reused; + } + + _meshes.Add(mesh); + return _meshes.Count - 1; + } + + /// Whether the mesh fetches its vertices through a storage buffer. + public bool IsSsbo(int meshId) => Get(meshId)?.Ssbo ?? false; + + /// + /// Fills an SSBO mesh's index buffer with the fixed quad pattern. + /// + /// GL keeps one shared static index buffer for every SSBO mesh, written once + /// with this pattern: each four consecutive vertices are a quad, drawn as the + /// triangles (0,1,2) and (0,2,3). Its chunk update path never uploads indices + /// on that route, so this fill is the only source of them here, as it is + /// there. + /// + private void FillQuadIndices(VulkanBuffer indices) + { + int count = (int)(indices.Size / sizeof(int)); + if (indices.Mapped == IntPtr.Zero) + { + if (_uploads == null) return; + var pattern = new int[count]; + fixed (int* source = pattern) + { + FillQuadPattern(source, count); + _uploads.UploadToBuffer(indices, 0, (IntPtr)source, (ulong)count * sizeof(int)); + } + return; + } + + FillQuadPattern((int*)indices.Mapped, count); + } + + private static void FillQuadPattern(int* destination, int count) + { + for (int i = 0; i + 5 < count; i += 6) + { + int quad = i / 6 * 4; + destination[i] = quad; + destination[i + 1] = quad + 1; + destination[i + 2] = quad + 2; + destination[i + 3] = quad; + destination[i + 4] = quad + 2; + destination[i + 5] = quad + 3; + } + } + + public VulkanBuffer? BufferOf(int meshId, int slot) + { + VulkanMesh? mesh = Get(meshId); + if (mesh == null) return null; + return slot < 0 ? mesh.Indices : mesh.Buffers[slot]; + } + + public IntPtr MappedPointer(int meshId, int slot) + { + VulkanMesh? mesh = Get(meshId); + if (mesh == null) return IntPtr.Zero; + + if (slot < 0) return mesh.Indices?.Mapped ?? IntPtr.Zero; + return slot < MaxBuffers ? mesh.Buffers[slot]?.Mapped ?? IntPtr.Zero : IntPtr.Zero; + } + + /// Writes bytes into a mesh buffer through its mapping. + /// + /// Copies data into one of a mesh's buffers at a byte offset. + /// + /// A write that cannot land - no such buffer, not mapped, or past the end - + /// is counted and traced rather than dropped in silence. GL would raise + /// GL_INVALID_VALUE for the same glBufferSubData; a quiet return here turned + /// a sizing mistake into "the terrain is simply not there", with nothing in + /// any log to say why. + /// + public void Write(int meshId, int slot, int byteOffset, IntPtr source, int byteCount) + { + if (source == IntPtr.Zero || byteCount <= 0) return; + + VulkanMesh? mesh = Get(meshId); + VulkanBuffer? buffer = mesh == null ? null : slot < 0 ? mesh.Indices : mesh.Buffers[slot]; + + string? problem = + mesh == null ? "no such mesh" : + buffer == null ? "mesh has no buffer in that slot" : + buffer.Mapped == IntPtr.Zero && _uploads == null ? "buffer is not host mapped" : + byteOffset < 0 ? "negative offset" : + (ulong)byteOffset + (ulong)byteCount > buffer.Size + ? "write ends past the buffer (" + buffer.Size + " bytes)" + : null; + + if (problem != null) + { + VulkanStats.NoteDroppedMeshWrite(); + if (RenderTrace.Enabled) + { + RenderTrace.Write("mesh write dropped: mesh " + meshId + " slot " + slot + + " offset " + byteOffset + " bytes " + byteCount + ": " + problem); + } + return; + } + + if (buffer!.Mapped == IntPtr.Zero) + { + // Device-local: staged and copied, never waited on. + _uploads!.UploadToBuffer(buffer, (ulong)byteOffset, source, (ulong)byteCount); + return; + } + + System.Buffer.MemoryCopy( + (void*)source, (void*)(buffer.Mapped + byteOffset), byteCount, byteCount); + } + + public void Delete(int meshId, FrameRing? ring = null) + { + VulkanMesh? mesh = Get(meshId); + if (mesh == null) return; + + _meshes[meshId] = null; + _freeIds.Push(meshId); + + if (ring != null) ring.DeferDeletion(mesh); + else mesh.Dispose(); + } + + // --------------------------------------------------------------------- draw + + /// Binds the mesh's vertex and index buffers. + public void Bind(CommandBuffer commandBuffer, VulkanMesh mesh) + { + Vk api = _context.Api; + + // A staged write to a buffer this frame command buffer already drew from + // has to go inline to keep GL's order; see UploadManager. + if (_uploads != null) + { + foreach (VulkanBuffer? buffer in mesh.Buffers) + { + if (buffer != null) _uploads.NoteUse(commandBuffer, buffer); + } + if (mesh.Indices != null) _uploads.NoteUse(commandBuffer, mesh.Indices); + } + + if (mesh.BindingOrder.Count > 0) + { + var buffers = new Buffer[mesh.BindingOrder.Count]; + var offsets = new ulong[mesh.BindingOrder.Count]; + for (int i = 0; i < mesh.BindingOrder.Count; i++) + { + buffers[i] = mesh.Buffers[mesh.BindingOrder[i]]!.Handle; + } + + fixed (Buffer* buffersPtr = buffers) + fixed (ulong* offsetsPtr = offsets) + { + api.CmdBindVertexBuffers(commandBuffer, 0, (uint)buffers.Length, buffersPtr, offsetsPtr); + } + } + + if (mesh.Indices != null) + { + // Always 32-bit: the game's index arrays are int[]. + api.CmdBindIndexBuffer(commandBuffer, mesh.Indices.Handle, 0, IndexType.Uint32); + } + } + + public void Draw(CommandBuffer commandBuffer, int meshId, int instanceCount = 1) + { + VulkanMesh? mesh = Get(meshId); + if (mesh == null || mesh.IndexCount == 0) return; + + Bind(commandBuffer, mesh); + _context.Api.CmdDrawIndexed(commandBuffer, (uint)mesh.IndexCount, (uint)instanceCount, 0, 0, 0); + } + + /// + /// The multidraw the chunk renderer issues once per pool, replacing + /// glMultiDrawElements. Records go through an indirect buffer. + /// + public void DrawMulti( + CommandBuffer commandBuffer, int meshId, + int[] indicesStarts, int[] indicesSizes, int groupCount, VulkanBuffer indirectScratch, + ulong indirectOffset) + { + VulkanMesh? mesh = Get(meshId); + if (mesh == null || groupCount <= 0) return; + + Bind(commandBuffer, mesh); + + if (indirectScratch.Mapped == IntPtr.Zero || indirectOffset >= indirectScratch.Size) return; + var commands = (DrawIndexedIndirectCommand*)(indirectScratch.Mapped + (nint)indirectOffset); + + int capacity = (int)((indirectScratch.Size - indirectOffset) / (ulong)sizeof(DrawIndexedIndirectCommand)); + int count = Math.Min(groupCount, capacity); + + if (count < groupCount && RenderTrace.Enabled) + { + RenderTrace.Write("mesh indirect draw clamped: mesh " + meshId + " groupCount " + groupCount + + " capacity " + capacity); + } + + WriteIndirectCommands(new Span(commands, count), indicesStarts, indicesSizes); + + _context.Api.CmdDrawIndexedIndirect(commandBuffer, indirectScratch.Handle, indirectOffset, (uint)count, + (uint)sizeof(DrawIndexedIndirectCommand)); + } + + internal static void WriteIndirectCommands( + Span commands, ReadOnlySpan indicesStarts, ReadOnlySpan indicesSizes) + { + for (int i = 0; i < commands.Length; i++) + { + // MeshDataPool passes GL's 64-bit pointer array in an int[]. Each + // offset occupies two words, unlike the tightly packed counts. + ulong byteOffset = (uint)indicesStarts[i * 2] | ((ulong)(uint)indicesStarts[i * 2 + 1] << 32); + commands[i] = new DrawIndexedIndirectCommand + { + IndexCount = (uint)indicesSizes[i], + InstanceCount = 1, + // GL takes a byte offset; Vulkan takes an index count. + FirstIndex = checked((uint)(byteOffset / sizeof(int))), + VertexOffset = 0, + FirstInstance = 0, + }; + } + } + + public int LayoutIdOf(int meshId) => Get(meshId)?.LayoutId ?? -1; + public VertexLayoutDescription LayoutOf(int layoutId) => _layouts.Get(layoutId); + public int LayoutCount => _layouts.Count; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + foreach (VulkanMesh? mesh in _meshes) mesh?.Dispose(); + _meshes.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Core/PipelineCache.cs b/Optimum.Render.Vulkan/Core/PipelineCache.cs new file mode 100644 index 00000000..2761296e --- /dev/null +++ b/Optimum.Render.Vulkan/Core/PipelineCache.cs @@ -0,0 +1,1127 @@ +using Optimum.Render.Vulkan.Shaders; +using System; +using System.Collections.Concurrent; +using System.Collections.Generic; +using System.Threading; +using Silk.NET.Core.Native; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Creates graphics pipelines on demand and remembers them. +/// +/// Vulkan wants pipeline state baked ahead of time; GL lets it change one call +/// before a draw. Bridging that is the job here: a draw states its fixed state +/// (NativePipelineDescription), and the first draw that needs a given +/// combination compiles a pipeline for it. Because Vulkan 1.3 makes viewport, +/// scissor, cull, front face, depth and stencil dynamic, the combinations that +/// remain are few - roughly a few hundred across the whole game - and after the +/// first minutes of play the cache stops growing. +/// +/// A driver-side backs it so that +/// even those first compiles are cheap on a second run. +/// +internal sealed unsafe class GraphicsPipelineCache : IDisposable +{ + private readonly VulkanContext _context; + private readonly Dictionary _pipelines = new(); + private readonly Silk.NET.Vulkan.PipelineCache _driverCache; + + /// + /// Serialises every host access to : creations against it, + /// merges into it and reads of its data. Merges require it (the destination is externally + /// synchronised), and serialising the reads and merges is the workaround for the AMD + /// reports of parallel creation corrupting cache data (docs/vulkan.md#caches §1, + /// "Design for this renderer" item 4). The background compiles themselves run against a + /// cache of their own, outside this lock. + /// + private readonly object _driverCacheLock = new(); + + private readonly DynamicState[] _dynamicStates; + private bool _disposed; + + /// How many pipelines have been compiled, for diagnostics. + public int Count => _pipelines.Count; + + /// How many lookups were served from the cache. + public long Hits { get; private set; } + + /// How many lookups had to compile. + public long Misses { get; private set; } + + private long _compiledSync; + private long _compiledAsync; + private long _prewarmedCount; + private long _warm; + private long _prewarmHits; + private long _drawsSkipped; + private long _queuedCompiles; + + /// Pipelines compiled on the calling thread. + public long CompiledSync => Interlocked.Read(ref _compiledSync); + + /// Pipelines a lookup asked for that the background worker compiled. + public long CompiledAsync => Interlocked.Read(ref _compiledAsync); + + /// Pipelines the worker compiled from the key log before any lookup asked. + public long Prewarmed => Interlocked.Read(ref _prewarmedCount); + + /// FAIL_ON_PIPELINE_COMPILE_REQUIRED creations the driver cache satisfied without a compile. + public long Warm => Interlocked.Read(ref _warm); + + /// First lookups of a key served by a prewarmed pipeline. + public long PrewarmHits => Interlocked.Read(ref _prewarmHits); + + /// Lookups that reported "not ready" (the draw was skipped). + public long DrawsSkipped => Interlocked.Read(ref _drawsSkipped); + + /// Jobs handed to the background worker for a lookup (prewarm jobs not included). + public long QueuedCompiles => Interlocked.Read(ref _queuedCompiles); + + private bool _asyncCompiles; + + /// + /// Whether may hand compiles to the background worker. Only takes + /// effect on a device with pipelineCreationCacheControl; off (the default) keeps every + /// lookup blocking, as always is. + /// + public bool AsyncCompiles + { + get => _asyncCompiles; + set => _asyncCompiles = value && _context.Capabilities.PipelineCreationCacheControl && !_workersStopped; + } + + /// Where every pipeline this cache hands out is recorded, for the next launch's prewarm. Null records nothing. + public PipelineKeyLog? KeyLog { get; set; } + + /// The settings hash key-log entries of this cache carry (). + public ulong SettingsHash { get; } + + /// The colour write tier every pipeline of this cache is built for. + public ColorWriteTier ColorWriteTier { get; } + + /// Everything Vulkan 1.3 core lets us change without a new pipeline. + private static readonly DynamicState[] CoreDynamicStates = + { + DynamicState.Viewport, + DynamicState.Scissor, + DynamicState.LineWidth, + DynamicState.CullMode, + DynamicState.FrontFace, + DynamicState.PrimitiveTopology, + DynamicState.DepthTestEnable, + DynamicState.DepthWriteEnable, + DynamicState.DepthCompareOp, + DynamicState.StencilTestEnable, + DynamicState.StencilOp, + DynamicState.StencilCompareMask, + DynamicState.StencilWriteMask, + DynamicState.StencilReference, + }; + + public GraphicsPipelineCache(VulkanContext context, byte[]? initialData = null) + : this(context, ColorWriteTier.PipelineKey, dynamicBlend: false, initialData) + { + } + + /// + /// A cache whose pipelines declare the colour write state of + /// dynamic (and the blend set, with on the mask tier). + /// Draws through it must then emit that state (VulkanDevice.ApplyDynamicState). + /// + public GraphicsPipelineCache(VulkanContext context, ColorWriteTier tier, bool dynamicBlend, byte[]? initialData = null) + { + _context = context; + ColorWriteTier = tier; + SettingsHash = PipelineKeyLog.SettingsHashFor(tier, dynamicBlend); + + var dynamicStates = new List(CoreDynamicStates); + if (tier == ColorWriteTier.DynamicEnable) dynamicStates.Add(DynamicState.ColorWriteEnableExt); + if (tier == ColorWriteTier.DynamicMask) + { + dynamicStates.Add(DynamicState.ColorWriteMaskExt); + if (dynamicBlend) + { + dynamicStates.Add(DynamicState.ColorBlendEnableExt); + dynamicStates.Add(DynamicState.ColorBlendEquationExt); + } + } + _dynamicStates = dynamicStates.ToArray(); + + // A rejected blob is not an error: the driver starts cold instead. Some + // drivers return an error rather than an empty cache for data they do not + // accept, so that case retries without it (docs/vulkan.md#caches §1). + if (initialData is { Length: > 0 } && TryCreateDriverCache(context, initialData, out _driverCache)) + { + SeedAccepted = true; + } + else + { + TryCreateDriverCache(context, null, out _driverCache); + } + } + + /// The driver's pipeline cache; compute pipelines compile through it too, so one file warms both. + public Silk.NET.Vulkan.PipelineCache DriverCache => _driverCache; + + /// The driver created its cache from the initial data rather than empty. + public bool SeedAccepted { get; } + + private static bool TryCreateDriverCache( + VulkanContext context, byte[]? initialData, out Silk.NET.Vulkan.PipelineCache cache) + { + fixed (byte* data = initialData) + { + // Never a non-null pointer with a zero size: one driver fails on exactly that. + var createInfo = new PipelineCacheCreateInfo + { + SType = StructureType.PipelineCacheCreateInfo, + InitialDataSize = (nuint)(initialData?.Length ?? 0), + PInitialData = initialData is { Length: > 0 } ? data : null, + }; + + if (context.Api.CreatePipelineCache(context.Device, &createInfo, null, out cache) == Result.Success) + { + return true; + } + cache = default; + return false; + } + } + + /// Everything a pipeline needs that is not already in the key. + internal sealed class PipelineRequest + { + public required ShaderProgramResources Program { get; init; } + public required VertexLayoutDescription VertexLayout { get; init; } + public required RenderTargetFormats Targets { get; init; } + public required AttachmentBlend[] Blend { get; init; } + public required PolygonMode PolygonMode { get; init; } + public required PrimitiveTopology Topology { get; init; } + } + + /// The pipeline for , compiled on this thread if it has to be. + public Pipeline Get(PipelineKey key, PipelineRequest request) + { + if (_pipelines.TryGetValue(key, out Pipeline existing)) + { + Hits++; + return existing; + } + + Misses++; + PipelineKeyLogEntry entry = PipelineKeyLogEntry.From(SettingsHash, request); + if (!TryAdoptPrewarmed((request.Program.ProgramId, entry.ContentId), out Pipeline pipeline)) + { + pipeline = CreateBlocking(request); + } + Store(key, pipeline, entry); + return pipeline; + } + + /// + /// The pipeline for if it can be had without compiling on this + /// thread; otherwise false, with the compile queued on the background worker, and the + /// caller skips its draw (Unreal's default for a PSO that is not ready, + /// docs/vulkan.md#caches §2). A key already compiling is not queued again. + /// Finished compiles become visible at . + /// + /// With off this never returns false. + /// + /// + /// ahead of any draw: the compile starts (or the driver cache serves it) + /// when a native system asks for its pipeline, and no draw is counted as skipped for it. + /// + public void Prepare(PipelineKey key, PipelineRequest request) + { + _preparing = true; + try + { + TryGet(key, request, out _); + } + finally + { + _preparing = false; + } + } + + private bool _preparing; + + public bool TryGet(PipelineKey key, PipelineRequest request, out Pipeline pipeline) + { + if (_pipelines.TryGetValue(key, out pipeline)) + { + Hits++; + return true; + } + + if (_pendingByKey.ContainsKey(key)) + { + NoteSkipped(); + return false; + } + + Misses++; + PipelineKeyLogEntry entry = PipelineKeyLogEntry.From(SettingsHash, request); + var id = (request.Program.ProgramId, entry.ContentId); + + if (TryAdoptPrewarmed(id, out pipeline)) + { + Store(key, pipeline, entry); + return true; + } + + // The same pipeline under another key, or a prewarm of it, is already on its way. + if (_pendingJobs.TryGetValue(id, out CompileJob? pending)) + { + // A prewarm no worker has reached yet never had a warm attempt: on a warm start + // the driver cache serves it here, and waiting for the queue would skip the draw. + if (TryTakeWarmFromQueuedPrewarm(pending, request, out pipeline)) + { + Store(key, pipeline, entry); + return true; + } + pending.DemandKeys.Add(key); + _pendingByKey[key] = pending; + Promote(pending); + NoteSkipped(); + return false; + } + + if (!AsyncCompiles || _failedJobs.Contains(id)) + { + pipeline = CreateBlocking(request); + Store(key, pipeline, entry); + return true; + } + + Result result; + lock (_driverCacheLock) + { + result = CreatePipeline(request, _driverCache, PipelineCreateFlags.CreateFailOnPipelineCompileRequiredBit, + out pipeline); + } + if (result == Result.Success) + { + Interlocked.Increment(ref _warm); + VulkanStats.NotePipelineWarm(); + Store(key, pipeline, entry); + return true; + } + if (result != Result.PipelineCompileRequired) + { + throw new InvalidOperationException("vkCreateGraphicsPipelines failed: " + result); + } + + var job = new CompileJob(id, request, entry, prewarm: false); + if (!TryEnqueue(job)) + { + // The worker is saturated; a draw is never left waiting on a queue it cannot join. + pipeline = CreateBlocking(request); + Store(key, pipeline, entry); + return true; + } + + Interlocked.Increment(ref _queuedCompiles); + job.DemandKeys.Add(key); + _pendingJobs[id] = job; + _pendingByKey[key] = job; + VulkanStats.NotePipelinesPending(_pendingJobs.Count); + NoteSkipped(); + return false; + } + + /// + /// The FAIL_ON_PIPELINE_COMPILE_REQUIRED attempt for a lookup whose pipeline is only + /// queued as a prewarm. On success the prewarm is dropped: removed from its queue, or, + /// when a worker took it meanwhile, cancelled so its result is destroyed at publication. + /// A prewarm already promoted by an earlier lookup had its attempt then and gets none. + /// + private bool TryTakeWarmFromQueuedPrewarm(CompileJob job, PipelineRequest request, out Pipeline pipeline) + { + pipeline = default; + lock (_queueLock) + { + if (!job.Prewarm || !job.Queued) return false; + } + + Result result; + lock (_driverCacheLock) + { + result = CreatePipeline(request, _driverCache, PipelineCreateFlags.CreateFailOnPipelineCompileRequiredBit, + out pipeline); + } + if (result == Result.PipelineCompileRequired) return false; + if (result != Result.Success) + { + throw new InvalidOperationException("vkCreateGraphicsPipelines failed: " + result); + } + + Interlocked.Increment(ref _warm); + VulkanStats.NotePipelineWarm(); + lock (_queueLock) + { + if (job.Queued) + { + job.Queued = false; + if (!_prewarmQueue.Remove(job)) _demandQueue.Remove(job); + } + else + { + job.Cancelled = true; + } + Forget(job); + } + VulkanStats.NotePipelinesPending(_pendingJobs.Count); + return true; + } + + private void NoteSkipped() + { + if (_preparing) return; + Interlocked.Increment(ref _drawsSkipped); + VulkanStats.NotePipelineDrawSkipped(); + } + + private Pipeline CreateBlocking(PipelineRequest request) + { + Result result; + Pipeline pipeline; + lock (_driverCacheLock) + { + result = CreatePipeline(request, _driverCache, 0, out pipeline); + } + if (result != Result.Success) + { + throw new InvalidOperationException("vkCreateGraphicsPipelines failed: " + result); + } + Interlocked.Increment(ref _compiledSync); + VulkanStats.NotePipelineCompiledSync(); + return pipeline; + } + + private void Store(PipelineKey key, Pipeline pipeline, PipelineKeyLogEntry entry) + { + _pipelines[key] = pipeline; + KeyLog?.Record(entry, DateTimeOffset.UtcNow.ToUnixTimeMilliseconds()); + } + + private bool TryAdoptPrewarmed((int ProgramId, UInt128 ContentId) id, out Pipeline pipeline) + { + if (!_prewarmed.Remove(id, out pipeline)) return false; + Interlocked.Increment(ref _prewarmHits); + return true; + } + + // ------------------------------------------------------------ background compiles + + /// A compile for the worker. Fields other than the result are touched by the render thread only. + private sealed class CompileJob + { + public CompileJob((int ProgramId, UInt128 ContentId) id, PipelineRequest request, PipelineKeyLogEntry entry, + bool prewarm) + { + Id = id; + Request = request; + Entry = entry; + Prewarm = prewarm; + } + + public (int ProgramId, UInt128 ContentId) Id { get; } + public PipelineRequest Request { get; } + public PipelineKeyLogEntry Entry { get; } + + /// Built from the key log; cleared when a lookup promotes it. Under the queue lock. + public bool Prewarm; + + /// Waiting in a queue rather than running or done. Under the queue lock. + public bool Queued; + + /// Its program was deleted; the result is destroyed rather than published. + public volatile bool Cancelled; + + /// The keys whose lookups wait for this pipeline. + public readonly List DemandKeys = new(); + + public Pipeline Pipeline; + public Result Status; + } + + /// Lookups first; each queue bounded so a burst cannot grow memory without limit. + internal const int DemandQueueCapacity = 256; + + internal const int PrewarmQueueCapacity = 4096; + + private readonly object _queueLock = new(); + private readonly LinkedList _demandQueue = new(); + private readonly LinkedList _prewarmQueue = new(); + private readonly List _inFlight = new(); + private readonly ConcurrentQueue _completed = new(); + private Thread[]? _workers; + private bool _stopping; + private bool _workersStopped; + private bool _holdForTests; + + /// + /// Tests only: while true the workers take no new job, so a test can observe a lookup + /// against a job that is still queued. Setting it false wakes them. + /// + internal bool HoldBackgroundCompilesForTests + { + set + { + lock (_queueLock) + { + _holdForTests = value; + Monitor.PulseAll(_queueLock); + } + } + } + + /// Render thread: every queued or running job by pipeline identity, and by the keys waiting on it. + private readonly Dictionary<(int ProgramId, UInt128 ContentId), CompileJob> _pendingJobs = new(); + private readonly Dictionary _pendingByKey = new(); + + /// Render thread: prewarmed pipelines no lookup has claimed yet. + private readonly Dictionary<(int ProgramId, UInt128 ContentId), Pipeline> _prewarmed = new(); + + /// Render thread: jobs whose compile failed; their next lookup compiles blocking and reports the error. + private readonly HashSet<(int ProgramId, UInt128 ContentId)> _failedJobs = new(); + + /// Compiles queued or running, prewarm included. + public int PendingCompiles => _pendingJobs.Count; + + /// Prewarmed pipelines published and not yet claimed by a lookup. + public int PrewarmedWaiting => _prewarmed.Count; + + private bool TryEnqueue(CompileJob job) + { + lock (_queueLock) + { + if (_stopping) return false; + LinkedList queue = job.Prewarm ? _prewarmQueue : _demandQueue; + int capacity = job.Prewarm ? PrewarmQueueCapacity : DemandQueueCapacity; + if (queue.Count >= capacity) return false; + queue.AddLast(job); + job.Queued = true; + _workers ??= StartWorkers(); + Monitor.Pulse(_queueLock); + return true; + } + } + + /// A lookup waits on a prewarm job still in its queue: it moves to the lookup queue. + private void Promote(CompileJob job) + { + lock (_queueLock) + { + if (!job.Prewarm) return; + job.Prewarm = false; + if (!job.Queued) return; + _prewarmQueue.Remove(job); + _demandQueue.AddLast(job); + } + } + + private Thread[] StartWorkers() + { + // One or two: compiles hold the main cache's lock only for the short warm attempt + // and the merge, so a second worker helps a cold start; more would compete with the + // game's own threads. + int count = Math.Clamp(Environment.ProcessorCount / 4, 1, 2); + var workers = new Thread[count]; + for (int i = 0; i < count; i++) + { + workers[i] = new Thread(WorkerLoop) + { + IsBackground = true, + Name = "optimum-pipeline-compile-" + i, + }; + workers[i].Start(); + } + return workers; + } + + private void WorkerLoop() + { + while (true) + { + CompileJob job; + lock (_queueLock) + { + while (!_stopping && (_holdForTests || (_demandQueue.Count == 0 && _prewarmQueue.Count == 0))) + { + Monitor.Wait(_queueLock); + } + if (_stopping) return; + LinkedList queue = _demandQueue.Count > 0 ? _demandQueue : _prewarmQueue; + job = queue.First!.Value; + queue.RemoveFirst(); + job.Queued = false; + _inFlight.Add(job); + } + + try + { + Compile(job); + } + finally + { + _completed.Enqueue(job); + lock (_queueLock) + { + _inFlight.Remove(job); + Monitor.PulseAll(_queueLock); + } + } + } + } + + private void Compile(CompileJob job) + { + Vk api = _context.Api; + + // The main cache may already hold it (a warm start): no compile, no merge. + lock (_driverCacheLock) + { + job.Status = CreatePipeline(job.Request, _driverCache, PipelineCreateFlags.CreateFailOnPipelineCompileRequiredBit, + out job.Pipeline); + } + if (job.Status == Result.Success) + { + Interlocked.Increment(ref _warm); + VulkanStats.NotePipelineWarm(); + NoteBuilt(job); + return; + } + if (job.Status != Result.PipelineCompileRequired) return; + + // The compile itself runs against a cache of this job's own, so the render thread's + // warm attempts never wait on it; the result is merged into the main cache after. + if (!TryCreateDriverCache(_context, null, out Silk.NET.Vulkan.PipelineCache local)) + { + job.Status = Result.ErrorInitializationFailed; + return; + } + try + { + job.Status = CreatePipeline(job.Request, local, 0, out job.Pipeline); + if (job.Status != Result.Success) return; + + lock (_driverCacheLock) + { + api.MergePipelineCaches(_context.Device, _driverCache, 1, &local); + } + + bool prewarm; + lock (_queueLock) prewarm = job.Prewarm; + if (!prewarm) + { + Interlocked.Increment(ref _compiledAsync); + VulkanStats.NotePipelineCompiledAsync(); + } + NoteBuilt(job); + } + finally + { + api.DestroyPipelineCache(_context.Device, local, null); + } + } + + /// Counts a prewarm job's pipeline once it exists, compiled or taken from the driver cache. + private void NoteBuilt(CompileJob job) + { + bool prewarm; + lock (_queueLock) prewarm = job.Prewarm; + if (!prewarm) return; + Interlocked.Increment(ref _prewarmedCount); + VulkanStats.NotePipelinePrewarmed(); + } + + /// + /// Render thread, at a safe point (frame start): makes the worker's finished pipelines + /// visible to lookups. Returns how many were published. + /// + public int PublishCompleted() + { + int published = 0; + while (_completed.TryDequeue(out CompileJob? job)) + { + // A job dropped early (Forget) may have been replaced under its id or keys by a + // newer one; only this job's own entries go. + Forget(job); + + if (job.Status != Result.Success || job.Pipeline.Handle == 0) + { + // A lookup waiting on it retries blocking and surfaces the error there. + if (!job.Cancelled) _failedJobs.Add(job.Id); + continue; + } + + if (job.Cancelled || _disposed) + { + _context.Api.DestroyPipeline(_context.Device, job.Pipeline, null); + continue; + } + + bool used = false; + foreach (PipelineKey key in job.DemandKeys) + { + if (_pipelines.ContainsKey(key)) continue; + Store(key, job.Pipeline, job.Entry); + used = true; + } + if (!used && (job.DemandKeys.Count > 0 || !_prewarmed.TryAdd(job.Id, job.Pipeline))) + { + _context.Api.DestroyPipeline(_context.Device, job.Pipeline, null); + continue; + } + published++; + } + VulkanStats.NotePipelinesPending(_pendingJobs.Count); + return published; + } + + /// + /// Render thread, when links: queues a background build of + /// every key-log entry recorded for a program with the same SPIR-V under this cache's + /// settings. Needs . Returns how many were queued. + /// + public int PrewarmFor(ShaderProgramResources program) + { + if (KeyLog == null || !AsyncCompiles) return 0; + + int queued = 0; + foreach (PipelineKeyLogEntry entry in KeyLog.Matching(SettingsHash, program.SourceHash)) + { + var id = (program.ProgramId, entry.ContentId); + if (_pendingJobs.ContainsKey(id) || _prewarmed.ContainsKey(id)) continue; + + var job = new CompileJob(id, entry.ToRequest(program), entry, prewarm: true); + if (!TryEnqueue(job)) break; + _pendingJobs[id] = job; + queued++; + } + VulkanStats.NotePipelinesPending(_pendingJobs.Count); + return queued; + } + + /// + /// Render thread, before is destroyed: drops its queued + /// compiles and waits for any the worker is running, so no compile ever reads a + /// destroyed shader module or layout. + /// + public void CancelProgram(ShaderProgramResources program) + { + // Every job of the program the render thread still tracks, including one the worker + // finished that waits in the completed queue for the next frame start: publishing + // that one would park a pipeline of a deleted program where no lookup can claim it. + foreach (CompileJob job in _pendingJobs.Values) + { + if (ReferenceEquals(job.Request.Program, program)) job.Cancelled = true; + } + + lock (_queueLock) + { + foreach (LinkedList queue in new[] { _demandQueue, _prewarmQueue }) + { + for (LinkedListNode? node = queue.First; node != null;) + { + LinkedListNode? next = node.Next; + if (ReferenceEquals(node.Value.Request.Program, program)) + { + node.Value.Cancelled = true; + node.Value.Queued = false; + queue.Remove(node); + Forget(node.Value); + } + node = next; + } + } + + while (true) + { + bool running = false; + foreach (CompileJob job in _inFlight) + { + if (!ReferenceEquals(job.Request.Program, program)) continue; + job.Cancelled = true; + running = true; + } + if (!running) break; + Monitor.Wait(_queueLock); + } + } + + var stale = new List<(int ProgramId, UInt128 ContentId)>(); + foreach ((int ProgramId, UInt128 ContentId) id in _prewarmed.Keys) + { + if (id.ProgramId == program.ProgramId) stale.Add(id); + } + foreach ((int ProgramId, UInt128 ContentId) id in stale) + { + _prewarmed.Remove(id, out Pipeline pipeline); + _context.Api.DestroyPipeline(_context.Device, pipeline, null); + } + VulkanStats.NotePipelinesPending(_pendingJobs.Count); + } + + private void Forget(CompileJob job) + { + if (_pendingJobs.TryGetValue(job.Id, out CompileJob? current) && ReferenceEquals(current, job)) + { + _pendingJobs.Remove(job.Id); + } + foreach (PipelineKey key in job.DemandKeys) + { + if (_pendingByKey.TryGetValue(key, out CompileJob? waiting) && ReferenceEquals(waiting, job)) + { + _pendingByKey.Remove(key); + } + } + } + + /// Waits until no compile is queued or running. Tests only; false on timeout. + internal bool WaitForBackgroundCompiles(TimeSpan timeout) + { + long deadline = Environment.TickCount64 + (long)timeout.TotalMilliseconds; + lock (_queueLock) + { + while (_demandQueue.Count > 0 || _prewarmQueue.Count > 0 || _inFlight.Count > 0) + { + long remaining = deadline - Environment.TickCount64; + if (remaining <= 0) return false; + Monitor.Wait(_queueLock, (int)Math.Min(remaining, int.MaxValue)); + } + } + return true; + } + + /// + /// Stops the workers after the job each is running, dropping queued ones. Lookups + /// compile blocking from then on. Before any program is destroyed at shutdown. + /// + public void StopBackgroundCompiles() + { + Thread[]? workers; + lock (_queueLock) + { + _stopping = true; + _workersStopped = true; + _asyncCompiles = false; + foreach (CompileJob job in _demandQueue) job.Cancelled = true; + foreach (CompileJob job in _prewarmQueue) job.Cancelled = true; + _demandQueue.Clear(); + _prewarmQueue.Clear(); + workers = _workers; + Monitor.PulseAll(_queueLock); + } + if (workers != null) + { + foreach (Thread worker in workers) worker.Join(); + } + while (_completed.TryDequeue(out CompileJob? job)) + { + if (job.Pipeline.Handle != 0) _context.Api.DestroyPipeline(_context.Device, job.Pipeline, null); + } + _pendingJobs.Clear(); + _pendingByKey.Clear(); + } + + private Result CreatePipeline(PipelineRequest request, Silk.NET.Vulkan.PipelineCache cache, + PipelineCreateFlags flags, out Pipeline pipeline) + { + Vk api = _context.Api; + byte* entryPoint = (byte*)SilkMarshal.StringToPtr("main"); + + // A native program's settings are specialization constants (docs/vulkan.md + // section 5); every stage gets the same map, and an id a module does not declare is ignored. + NativeSpecialization? specialization = request.Program.Specialization; + SpecializationInfo* specializationInfo = null; + if (specialization != null && specialization.Entries.Length > 0) + { + nuint entryBytes = (nuint)(sizeof(SpecializationMapEntry) * specialization.Entries.Length); + var map = (SpecializationMapEntry*)System.Runtime.InteropServices.NativeMemory.Alloc(entryBytes); + for (int i = 0; i < specialization.Entries.Length; i++) + { + NativeSpecialization.Entry entry = specialization.Entries[i]; + map[i] = new SpecializationMapEntry { ConstantID = entry.Id, Offset = entry.Offset, Size = entry.Size }; + } + var data = (byte*)System.Runtime.InteropServices.NativeMemory.Alloc((nuint)Math.Max(specialization.Data.Length, 1)); + specialization.Data.AsSpan().CopyTo(new Span(data, specialization.Data.Length)); + specializationInfo = (SpecializationInfo*)System.Runtime.InteropServices.NativeMemory.Alloc((nuint)sizeof(SpecializationInfo)); + *specializationInfo = new SpecializationInfo + { + MapEntryCount = (uint)specialization.Entries.Length, + PMapEntries = map, + DataSize = (nuint)specialization.Data.Length, + PData = data, + }; + } + + var stages = new List(); + foreach (KeyValuePair module in request.Program.Modules) + { + stages.Add(new PipelineShaderStageCreateInfo + { + SType = StructureType.PipelineShaderStageCreateInfo, + Stage = module.Key switch + { + EnumShaderType.VertexShader => ShaderStageFlags.VertexBit, + EnumShaderType.FragmentShader => ShaderStageFlags.FragmentBit, + _ => ShaderStageFlags.GeometryBit, + }, + Module = module.Value, + PName = entryPoint, + PSpecializationInfo = specializationInfo, + }); + } + + var bindings = new VertexInputBindingDescription[request.VertexLayout.Bindings.Length]; + for (int i = 0; i < bindings.Length; i++) + { + VertexBinding binding = request.VertexLayout.Bindings[i]; + bindings[i] = new VertexInputBindingDescription + { + Binding = binding.Binding, + Stride = binding.Stride, + InputRate = binding.PerInstance ? VertexInputRate.Instance : VertexInputRate.Vertex, + }; + } + + var attributes = new VertexInputAttributeDescription[request.VertexLayout.Attributes.Length]; + for (int i = 0; i < attributes.Length; i++) + { + VertexAttribute attribute = request.VertexLayout.Attributes[i]; + attributes[i] = new VertexInputAttributeDescription + { + Location = attribute.Location, + Binding = attribute.Binding, + Format = attribute.Format, + Offset = attribute.Offset, + }; + } + + var blendAttachments = new PipelineColorBlendAttachmentState[request.Targets.ColorFormats.Length]; + for (int i = 0; i < blendAttachments.Length; i++) + { + AttachmentBlend blend = i < request.Blend.Length ? request.Blend[i] : AttachmentBlend.Default; + // An enabled attachment the fragment shader never stores to keeps + // its contents, as it does on GL; Vulkan would write undefined + // values (validation: "Output variable was never written to"). + ColorComponentFlags writeMask = request.Program.Interface.WrittenFragmentOutputs.Contains(i) + ? blend.WriteMask + : 0; + blendAttachments[i] = new PipelineColorBlendAttachmentState + { + BlendEnable = blend.Enabled, + SrcColorBlendFactor = blend.SrcColor, + DstColorBlendFactor = blend.DstColor, + ColorBlendOp = blend.ColorOp, + SrcAlphaBlendFactor = blend.SrcAlpha, + DstAlphaBlendFactor = blend.DstAlpha, + AlphaBlendOp = blend.AlphaOp, + ColorWriteMask = writeMask, + }; + } + + // Everything Vulkan lets us change without a new pipeline, plus the + // colour write state of the tier. Keeping this list wide is what keeps + // the cache small. + DynamicState[] dynamicStates = _dynamicStates; + + try + { + fixed (PipelineShaderStageCreateInfo* stagesPtr = stages.ToArray()) + fixed (VertexInputBindingDescription* bindingsPtr = bindings) + fixed (VertexInputAttributeDescription* attributesPtr = attributes) + fixed (PipelineColorBlendAttachmentState* blendPtr = blendAttachments) + fixed (DynamicState* dynamicPtr = dynamicStates) + fixed (Format* colorFormatsPtr = request.Targets.ColorFormats) + { + var vertexInput = new PipelineVertexInputStateCreateInfo + { + SType = StructureType.PipelineVertexInputStateCreateInfo, + VertexBindingDescriptionCount = (uint)bindings.Length, + PVertexBindingDescriptions = bindings.Length == 0 ? null : bindingsPtr, + VertexAttributeDescriptionCount = (uint)attributes.Length, + PVertexAttributeDescriptions = attributes.Length == 0 ? null : attributesPtr, + }; + + var inputAssembly = new PipelineInputAssemblyStateCreateInfo + { + SType = StructureType.PipelineInputAssemblyStateCreateInfo, + Topology = request.Topology, + }; + + var viewportState = new PipelineViewportStateCreateInfo + { + SType = StructureType.PipelineViewportStateCreateInfo, + ViewportCount = 1, + ScissorCount = 1, + }; + + var rasterizer = new PipelineRasterizationStateCreateInfo + { + SType = StructureType.PipelineRasterizationStateCreateInfo, + PolygonMode = request.PolygonMode, + // Cull mode and front face are dynamic; these are placeholders. + CullMode = CullModeFlags.None, + FrontFace = RenderLimits.FrontFace, + LineWidth = 1.0f, + }; + + var multisample = new PipelineMultisampleStateCreateInfo + { + SType = StructureType.PipelineMultisampleStateCreateInfo, + RasterizationSamples = SampleCountFlags.Count1Bit, + }; + + var depthStencil = new PipelineDepthStencilStateCreateInfo + { + SType = StructureType.PipelineDepthStencilStateCreateInfo, + DepthTestEnable = false, + DepthWriteEnable = true, + DepthCompareOp = CompareOp.Less, + StencilTestEnable = false, + }; + + var colorBlend = new PipelineColorBlendStateCreateInfo + { + SType = StructureType.PipelineColorBlendStateCreateInfo, + AttachmentCount = (uint)blendAttachments.Length, + PAttachments = blendAttachments.Length == 0 ? null : blendPtr, + }; + + var dynamicState = new PipelineDynamicStateCreateInfo + { + SType = StructureType.PipelineDynamicStateCreateInfo, + DynamicStateCount = (uint)dynamicStates.Length, + PDynamicStates = dynamicPtr, + }; + + // Dynamic rendering names the attachment formats here, so there + // is no render pass or framebuffer object anywhere in the design. + var renderingInfo = new PipelineRenderingCreateInfo + { + SType = StructureType.PipelineRenderingCreateInfo, + ColorAttachmentCount = (uint)request.Targets.ColorFormats.Length, + PColorAttachmentFormats = request.Targets.ColorFormats.Length == 0 ? null : colorFormatsPtr, + DepthAttachmentFormat = request.Targets.DepthFormat, + }; + + var createInfo = new GraphicsPipelineCreateInfo + { + SType = StructureType.GraphicsPipelineCreateInfo, + // FAIL_ON_PIPELINE_COMPILE_REQUIRED for a warm attempt: the driver + // returns VK_PIPELINE_COMPILE_REQUIRED instead of compiling. + Flags = flags, + PNext = &renderingInfo, + StageCount = (uint)stages.Count, + PStages = stagesPtr, + PVertexInputState = &vertexInput, + PInputAssemblyState = &inputAssembly, + PViewportState = &viewportState, + PRasterizationState = &rasterizer, + PMultisampleState = &multisample, + PDepthStencilState = &depthStencil, + PColorBlendState = &colorBlend, + PDynamicState = &dynamicState, + Layout = request.Program.PipelineLayout, + }; + + Result result = api.CreateGraphicsPipelines( + _context.Device, cache, 1, &createInfo, null, out pipeline); + if (result != Result.Success) pipeline = default; + return result; + } + } + finally + { + SilkMarshal.Free((nint)entryPoint); + if (specializationInfo != null) + { + System.Runtime.InteropServices.NativeMemory.Free(specializationInfo->PMapEntries); + System.Runtime.InteropServices.NativeMemory.Free(specializationInfo->PData); + System.Runtime.InteropServices.NativeMemory.Free(specializationInfo); + } + } + } + + /// + /// The driver's cache blob, to be written next to the SPIR-V cache so the + /// next run starts warm (). + /// + public byte[] SerializeDriverCache() + { + if (_driverCache.Handle == 0) return Array.Empty(); + lock (_driverCacheLock) return SerializeDriverCacheLocked(); + } + + /// The serialised driver cache size in bytes, without copying it. Safe from any thread. + public long DriverCacheSize() + { + if (_driverCache.Handle == 0) return 0; + lock (_driverCacheLock) + { + nuint size = 0; + return _context.Api.GetPipelineCacheData(_context.Device, _driverCache, ref size, null) == Result.Success + ? (long)size + : 0; + } + } + + private byte[] SerializeDriverCacheLocked() + { + // The cache can grow between the size query and the fetch while another + // thread creates a pipeline; VK_INCOMPLETE then means "ask again". + for (int attempt = 0; attempt < 4; attempt++) + { + nuint size = 0; + if (_context.Api.GetPipelineCacheData(_context.Device, _driverCache, ref size, null) != Result.Success || + size == 0) + { + return Array.Empty(); + } + + var data = new byte[(int)size]; + Result result; + fixed (byte* dataPtr = data) + { + result = _context.Api.GetPipelineCacheData(_context.Device, _driverCache, ref size, dataPtr); + } + if (result == Result.Success) return size == (nuint)data.Length ? data : data[..(int)size]; + if (result != Result.Incomplete) break; + } + return Array.Empty(); + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + StopBackgroundCompiles(); + + Vk api = _context.Api; + // One pipeline can serve several keys (equal content under different interned ids). + var destroyed = new HashSet(); + foreach (Pipeline pipeline in _pipelines.Values) + { + if (destroyed.Add(pipeline.Handle)) api.DestroyPipeline(_context.Device, pipeline, null); + } + _pipelines.Clear(); + foreach (Pipeline pipeline in _prewarmed.Values) + { + if (destroyed.Add(pipeline.Handle)) api.DestroyPipeline(_context.Device, pipeline, null); + } + _prewarmed.Clear(); + + if (_driverCache.Handle != 0) + { + api.DestroyPipelineCache(_context.Device, _driverCache, null); + } + } +} diff --git a/Optimum.Render.Vulkan/Core/PipelineCachePersistence.cs b/Optimum.Render.Vulkan/Core/PipelineCachePersistence.cs new file mode 100644 index 00000000..98d92a9b --- /dev/null +++ b/Optimum.Render.Vulkan/Core/PipelineCachePersistence.cs @@ -0,0 +1,361 @@ +using System; +using System.Diagnostics; +using System.Threading; +using System.Threading.Tasks; +using System.IO; +using System.Security.Cryptography; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Decides when the driver's pipeline cache has grown enough to be worth writing out +/// before shutdown. +/// +/// A session that crashes, or is killed, loses everything a shutdown-only save would +/// have written. Godot saves from a worker when the cache has grown by some megabytes +/// rather than on a timer (docs/vulkan.md#caches §1, godot#76348); this is that +/// rule, with the size sampled at most once per interval because the size query goes +/// through the driver. +/// +internal sealed class PipelineCacheGrowthTrigger +{ + public const long DefaultThresholdBytes = 8L * 1024 * 1024; + + public static readonly TimeSpan DefaultInterval = TimeSpan.FromSeconds(10); + + private readonly long _thresholdBytes; + private readonly long _intervalTicks; + private long _lastSampleTimestamp; + private long _baselineBytes; + + /// What is on disk already: the seed the cache was created from, or 0. + public PipelineCacheGrowthTrigger(long thresholdBytes, TimeSpan interval, long baselineBytes) + { + _thresholdBytes = Math.Max(1, thresholdBytes); + _intervalTicks = (long)(interval.TotalSeconds * Stopwatch.Frequency); + _baselineBytes = baselineBytes; + } + + public long BaselineBytes => Interlocked.Read(ref _baselineBytes); + + /// + /// Whether a size sample is due at (Stopwatch ticks). The + /// first call only arms the clock; afterwards at most one sample per interval. + /// + public bool SampleDue(long timestamp) + { + if (_lastSampleTimestamp == 0) + { + _lastSampleTimestamp = timestamp; + return false; + } + if (timestamp - _lastSampleTimestamp < _intervalTicks) return false; + _lastSampleTimestamp = timestamp; + return true; + } + + /// The cache is at least the threshold larger than what was last written. + public bool GrewEnough(long currentBytes) => currentBytes - BaselineBytes >= _thresholdBytes; + + public void NoteSaved(long bytes) => Interlocked.Exchange(ref _baselineBytes, bytes); +} + +/// +/// The pipeline cache and pipeline-key log files of one GPU, and the saves that write them. +/// +/// The render thread only calls , a timestamp comparison; the size query, +/// the serialisation and the file writes run on a thread-pool task, one at a time. The +/// shutdown save stays: waits for a running save and writes +/// both files once more. +/// +internal sealed class PipelineCachePersistence +{ + private readonly PipelineCacheIdentity _identity; + private readonly PipelineCacheGrowthTrigger _trigger; + private Task? _pending; + private long _saves; + + public string CachePath { get; } + public string KeyLogPath { get; } + public PipelineKeyLog KeyLog { get; } + + /// Where a failed write is reported (the validation mirror in the device). + public Action? Log { get; set; } + + /// Driver cache files written, opportunistic and shutdown saves together. + public long Saves => Interlocked.Read(ref _saves); + + private PipelineCachePersistence(string cachePath, string keyLogPath, PipelineCacheIdentity identity, + PipelineKeyLog keyLog, PipelineCacheGrowthTrigger trigger) + { + CachePath = cachePath; + KeyLogPath = keyLogPath; + _identity = identity; + KeyLog = keyLog; + _trigger = trigger; + } + + /// Loads both files for ; is the usable driver blob or null. + public static PipelineCachePersistence Open(string cacheRoot, PipelineCacheIdentity identity, out byte[]? seed, + long thresholdBytes = PipelineCacheGrowthTrigger.DefaultThresholdBytes, TimeSpan? interval = null) + { + string cachePath = PipelineCacheFile.PathFor(cacheRoot, identity); + string keyLogPath = PipelineKeyLog.PathFor(cacheRoot, identity); + seed = PipelineCacheFile.Load(cachePath, identity); + return new PipelineCachePersistence(cachePath, keyLogPath, identity, PipelineKeyLog.Load(keyLogPath), + new PipelineCacheGrowthTrigger(thresholdBytes, interval ?? PipelineCacheGrowthTrigger.DefaultInterval, + seed?.Length ?? 0)); + } + + /// + /// Render thread, once per frame: starts a background save pass when a sample is due + /// and no pass is running. True when one started. + /// + public bool Tick(GraphicsPipelineCache cache, long timestamp) + { + if (_pending is { IsCompleted: false }) return false; + if (!_trigger.SampleDue(timestamp)) return false; + _pending = Task.Run(() => BackgroundPass(cache)); + return true; + } + + /// Waits for a running background pass. Before the cache is disposed. + public void WaitForPendingSave() + { + Task? pending = _pending; + if (pending == null) return; + try + { + pending.Wait(); + } + catch (AggregateException error) + { + Log?.Invoke("--- pipeline cache background save failed: " + error.InnerException?.Message); + } + } + + public void SaveAtShutdown(GraphicsPipelineCache cache) + { + WaitForPendingSave(); + SaveDriverCache(cache); + if (KeyLog.HasUnsavedChanges && !KeyLog.Save(KeyLogPath)) + { + Log?.Invoke("--- pipeline key log not saved to " + KeyLogPath); + } + } + + private void BackgroundPass(GraphicsPipelineCache cache) + { + long size = cache.DriverCacheSize(); + VulkanStats.NotePipelineCacheBytes(size); + if (_trigger.GrewEnough(size)) SaveDriverCache(cache); + if (KeyLog.HasUnsavedChanges && !KeyLog.Save(KeyLogPath)) + { + Log?.Invoke("--- pipeline key log not saved to " + KeyLogPath); + } + } + + private void SaveDriverCache(GraphicsPipelineCache cache) + { + byte[] blob = cache.SerializeDriverCache(); + if (blob.Length == 0) return; + VulkanStats.NotePipelineCacheBytes(blob.Length); + if (!PipelineCacheFile.Save(CachePath, blob, _identity)) + { + Log?.Invoke("--- pipeline cache not saved to " + CachePath); + return; + } + _trigger.NoteSaved(blob.Length); + Interlocked.Increment(ref _saves); + VulkanStats.NotePipelineCacheSave(); + } +} + +/// The device and driver a pipeline cache blob was produced by. +internal readonly record struct PipelineCacheIdentity(uint VendorId, uint DeviceId, uint DriverVersion, byte[] Uuid) +{ + public static PipelineCacheIdentity Of(VulkanCapabilities capabilities) => new( + capabilities.VendorId, capabilities.DeviceId, capabilities.DriverVersion, capabilities.PipelineCacheUuid); + + /// One file per GPU, so switching between two GPUs keeps both caches warm. + public string FileName => $"{VendorId:x4}-{DeviceId:x4}-{Convert.ToHexStringLower(Uuid)}.bin"; +} + +/// +/// The driver's pipeline cache blob on disk, wrapped so that it is only ever handed +/// back to the driver that wrote it, whole. +/// +/// The spec says a driver must start empty when the blob's own header does not match +/// it, but drivers have been seen to skip that check and crash after a driver update, +/// to keep the UUID across incompatible builds, and to fail on a zero-length blob; +/// files have been seen truncated, zero-filled and empty. So the wrapper records +/// the vendor, device, driver version, pointer size and UUID alongside a SHA-256 of +/// the blob. Every field and the blob's own Vulkan header are checked before loading, +/// and anything that fails is treated as no cache at all. +/// Design and sources: docs/vulkan.md#caches §1 and "Design for this renderer" item 2. +/// +internal static class PipelineCacheFile +{ + /// "OPLC", little-endian. + internal const uint FileMagic = 0x434C504F; + + internal const uint FormatVersion = 1; + + /// Magic, version, data size, SHA-256, vendor, device, driver version, pointer size, UUID. + internal const int HeaderSize = 4 + 4 + 8 + 32 + 4 + 4 + 4 + 4 + 16; + + /// VkPipelineCacheHeaderVersionOne: size, version, vendor, device, UUID. + private const int VulkanHeaderSize = 32; + + private const uint VulkanHeaderVersionOne = 1; + + public static string PathFor(string cacheRoot, PipelineCacheIdentity identity) => + Path.Combine(cacheRoot, "pipeline", identity.FileName); + + /// The blob stored for , or null when there is none it can use. + public static byte[]? Load(string path, PipelineCacheIdentity identity) + { + byte[]? file = CacheFileWriter.TryReadAll(path); + return file == null ? null : Unwrap(file, identity); + } + + /// Stores a blob; false when it was empty, not a pipeline cache, or could not be written. + public static bool Save(string path, byte[] data, PipelineCacheIdentity identity) + { + if (!HasMatchingVulkanHeader(data, identity)) return false; + return CacheFileWriter.WriteAtomically(path, Wrap(data, identity)); + } + + internal static byte[] Wrap(byte[] data, PipelineCacheIdentity identity) + { + var file = new byte[HeaderSize + data.Length]; + Span header = file.AsSpan(0, HeaderSize); + BitConverter.TryWriteBytes(header[0..], FileMagic); + BitConverter.TryWriteBytes(header[4..], FormatVersion); + BitConverter.TryWriteBytes(header[8..], (ulong)data.Length); + SHA256.HashData(data, header.Slice(16, 32)); + BitConverter.TryWriteBytes(header[48..], identity.VendorId); + BitConverter.TryWriteBytes(header[52..], identity.DeviceId); + BitConverter.TryWriteBytes(header[56..], identity.DriverVersion); + BitConverter.TryWriteBytes(header[60..], (uint)IntPtr.Size); + identity.Uuid.AsSpan(0, 16).CopyTo(header[64..]); + data.CopyTo(file, HeaderSize); + return file; + } + + /// The blob inside a file written for exactly , or null. + internal static byte[]? Unwrap(byte[] file, PipelineCacheIdentity identity) + { + if (file.Length <= HeaderSize) return null; + ReadOnlySpan header = file.AsSpan(0, HeaderSize); + if (BitConverter.ToUInt32(header[0..]) != FileMagic) return null; + if (BitConverter.ToUInt32(header[4..]) != FormatVersion) return null; + if (BitConverter.ToUInt64(header[8..]) != (ulong)(file.Length - HeaderSize)) return null; + if (BitConverter.ToUInt32(header[48..]) != identity.VendorId) return null; + if (BitConverter.ToUInt32(header[52..]) != identity.DeviceId) return null; + if (BitConverter.ToUInt32(header[56..]) != identity.DriverVersion) return null; + if (BitConverter.ToUInt32(header[60..]) != (uint)IntPtr.Size) return null; + if (!header.Slice(64, 16).SequenceEqual(identity.Uuid)) return null; + + ReadOnlySpan data = file.AsSpan(HeaderSize); + Span hash = stackalloc byte[32]; + SHA256.HashData(data, hash); + if (!hash.SequenceEqual(header.Slice(16, 32))) return null; + + byte[] blob = data.ToArray(); + return HasMatchingVulkanHeader(blob, identity) ? blob : null; + } + + /// The blob's own VkPipelineCacheHeaderVersionOne names this device. + internal static bool HasMatchingVulkanHeader(byte[] data, PipelineCacheIdentity identity) + { + if (data.Length < VulkanHeaderSize || identity.Uuid is not { Length: 16 }) return false; + ReadOnlySpan header = data; + return BitConverter.ToUInt32(header[0..]) >= VulkanHeaderSize + && BitConverter.ToUInt32(header[4..]) == VulkanHeaderVersionOne + && BitConverter.ToUInt32(header[8..]) == identity.VendorId + && BitConverter.ToUInt32(header[12..]) == identity.DeviceId + && header.Slice(16, 16).SequenceEqual(identity.Uuid); + } +} + +/// +/// Replaces a cache file without a reader ever seeing half of it. +/// +/// The bytes go to a temporary file unique to this process and call, which is then +/// moved over the destination. Two game instances saving at once lose one of the +/// two writes, never corrupt the file. On Windows a virus scanner briefly holds +/// newly written files open, which makes the move fail; the move is retried with +/// a short backoff before the write is given up (docs/vulkan.md#caches §7). +/// +internal static class CacheFileWriter +{ + internal const int MoveAttempts = 5; + + /// Writes to ; false when it could not. + public static bool WriteAtomically(string path, ReadOnlySpan bytes) => + WriteAtomically(path, bytes, static (from, to) => File.Move(from, to, overwrite: true), Thread.Sleep); + + /// + /// The same, with the replace step and the backoff sleep supplied: tests stand in for a + /// scanner holding the new file open. moves its first argument + /// over its second; takes milliseconds. + /// + internal static bool WriteAtomically(string path, ReadOnlySpan bytes, Action replace, + Action sleep) + { + string temporary = path + "." + Environment.ProcessId + "." + Guid.NewGuid().ToString("N") + ".tmp"; + try + { + Directory.CreateDirectory(Path.GetDirectoryName(path)!); + using (var stream = new FileStream(temporary, FileMode.CreateNew, FileAccess.Write, FileShare.None)) + { + stream.Write(bytes); + } + + for (int attempt = 1; ; attempt++) + { + try + { + replace(temporary, path); + return true; + } + catch (Exception error) when (IsTransient(error) && attempt < MoveAttempts) + { + sleep(10 << attempt); + } + } + } + catch (Exception error) when (IsTransient(error)) + { + TryDelete(temporary); + return false; + } + } + + /// The whole file, or null when it is missing or unreadable. + public static byte[]? TryReadAll(string path) + { + try + { + return File.Exists(path) ? File.ReadAllBytes(path) : null; + } + catch (Exception error) when (IsTransient(error)) + { + return null; + } + } + + private static bool IsTransient(Exception error) => error is IOException or UnauthorizedAccessException; + + private static void TryDelete(string path) + { + try + { + File.Delete(path); + } + catch (Exception error) when (IsTransient(error)) + { + } + } +} diff --git a/Optimum.Render.Vulkan/Core/PipelineKeyLog.cs b/Optimum.Render.Vulkan/Core/PipelineKeyLog.cs new file mode 100644 index 00000000..f1ad3633 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/PipelineKeyLog.cs @@ -0,0 +1,395 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Security.Cryptography; +using System.Text; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// One pipeline the renderer built, described by what decides it rather than by the +/// session's interned ids. +/// +/// A names its program, vertex layout, target formats and +/// blend set by ids handed out in first-use order, which mean nothing in the next +/// launch. This entry carries the content those ids stood for: the program's SPIR-V +/// hash (, which already has the +/// program's defines resolved), the vertex layout, the attachment formats, the blend +/// state per attachment, polygon mode and topology, plus the settings hash of the +/// device-wide state that shapes every pipeline (). +/// Design: docs/vulkan.md#caches "Design for this renderer" item 5. +/// +internal sealed class PipelineKeyLogEntry +{ + /// More than any attachment or attribute count Vulkan allows; a larger count is a damaged file. + private const int MaxArrayLength = 64; + + public required ulong SettingsHash { get; init; } + public required UInt128 ProgramHash { get; init; } + public required VertexBinding[] Bindings { get; init; } + public required VertexAttribute[] Attributes { get; init; } + public required Format[] ColorFormats { get; init; } + public required Format DepthFormat { get; init; } + + /// Exactly one per colour format: the pipeline builder pads a short blend array with the default. + public required AttachmentBlend[] Blend { get; init; } + + public required PolygonMode PolygonMode { get; init; } + public required PrimitiveTopology Topology { get; init; } + + /// Unix milliseconds of the last session that used this pipeline; the LRU order. + public long LastSeenUnixMs { get; set; } + + private UInt128? _contentId; + + /// A hash of everything above except : equal ids build equal pipelines. + public UInt128 ContentId => _contentId ??= ComputeContentId(); + + public static PipelineKeyLogEntry From(ulong settingsHash, GraphicsPipelineCache.PipelineRequest request) + { + Format[] colors = request.Targets.ColorFormats; + var blend = new AttachmentBlend[colors.Length]; + for (int i = 0; i < blend.Length; i++) + { + blend[i] = i < request.Blend.Length ? request.Blend[i] : AttachmentBlend.Default; + } + + return new PipelineKeyLogEntry + { + SettingsHash = settingsHash, + ProgramHash = request.Program.SourceHash, + Bindings = (VertexBinding[])request.VertexLayout.Bindings.Clone(), + Attributes = (VertexAttribute[])request.VertexLayout.Attributes.Clone(), + ColorFormats = (Format[])colors.Clone(), + DepthFormat = request.Targets.DepthFormat, + Blend = blend, + PolygonMode = request.PolygonMode, + Topology = request.Topology, + }; + } + + /// The request that rebuilds this pipeline for , whose hash must match. + public GraphicsPipelineCache.PipelineRequest ToRequest(ShaderProgramResources program) => new() + { + Program = program, + VertexLayout = new VertexLayoutDescription(Bindings, Attributes), + Targets = new RenderTargetFormats(ColorFormats, DepthFormat), + Blend = Blend, + PolygonMode = PolygonMode, + Topology = Topology, + }; + + private UInt128 ComputeContentId() + { + using var stream = new MemoryStream(); + using (var writer = new BinaryWriter(stream, Encoding.UTF8, leaveOpen: true)) + { + WriteContent(writer); + } + Span hash = stackalloc byte[32]; + SHA256.HashData(stream.GetBuffer().AsSpan(0, (int)stream.Length), hash); + return new UInt128(BitConverter.ToUInt64(hash[..8]), BitConverter.ToUInt64(hash.Slice(8, 8))); + } + + internal void WriteContent(BinaryWriter writer) + { + writer.Write(SettingsHash); + writer.Write((ulong)(ProgramHash >> 64)); + writer.Write((ulong)ProgramHash); + + writer.Write(Bindings.Length); + foreach (VertexBinding binding in Bindings) + { + writer.Write(binding.Binding); + writer.Write(binding.Stride); + writer.Write(binding.PerInstance); + } + + writer.Write(Attributes.Length); + foreach (VertexAttribute attribute in Attributes) + { + writer.Write(attribute.Location); + writer.Write(attribute.Binding); + writer.Write((int)attribute.Format); + writer.Write(attribute.Offset); + } + + writer.Write(ColorFormats.Length); + foreach (Format format in ColorFormats) writer.Write((int)format); + writer.Write((int)DepthFormat); + + writer.Write(Blend.Length); + foreach (AttachmentBlend blend in Blend) + { + writer.Write(blend.Enabled); + writer.Write((int)blend.SrcColor); + writer.Write((int)blend.DstColor); + writer.Write((int)blend.ColorOp); + writer.Write((int)blend.SrcAlpha); + writer.Write((int)blend.DstAlpha); + writer.Write((int)blend.AlphaOp); + writer.Write((uint)blend.WriteMask); + } + + writer.Write((int)PolygonMode); + writer.Write((int)Topology); + } + + /// One entry's content, or null when a count is out of range. Truncation throws EndOfStreamException. + internal static PipelineKeyLogEntry? ReadContent(BinaryReader reader) + { + ulong settingsHash = reader.ReadUInt64(); + ulong programHigh = reader.ReadUInt64(); + ulong programLow = reader.ReadUInt64(); + + int bindingCount = reader.ReadInt32(); + if ((uint)bindingCount > MaxArrayLength) return null; + var bindings = new VertexBinding[bindingCount]; + for (int i = 0; i < bindingCount; i++) + { + bindings[i] = new VertexBinding(reader.ReadUInt32(), reader.ReadUInt32(), reader.ReadBoolean()); + } + + int attributeCount = reader.ReadInt32(); + if ((uint)attributeCount > MaxArrayLength) return null; + var attributes = new VertexAttribute[attributeCount]; + for (int i = 0; i < attributeCount; i++) + { + attributes[i] = new VertexAttribute(reader.ReadUInt32(), reader.ReadUInt32(), (Format)reader.ReadInt32(), + reader.ReadUInt32()); + } + + int colorCount = reader.ReadInt32(); + if ((uint)colorCount > MaxArrayLength) return null; + var colors = new Format[colorCount]; + for (int i = 0; i < colorCount; i++) colors[i] = (Format)reader.ReadInt32(); + var depth = (Format)reader.ReadInt32(); + + int blendCount = reader.ReadInt32(); + if (blendCount != colorCount) return null; + var blend = new AttachmentBlend[blendCount]; + for (int i = 0; i < blendCount; i++) + { + blend[i] = new AttachmentBlend + { + Enabled = reader.ReadBoolean(), + SrcColor = (BlendFactor)reader.ReadInt32(), + DstColor = (BlendFactor)reader.ReadInt32(), + ColorOp = (BlendOp)reader.ReadInt32(), + SrcAlpha = (BlendFactor)reader.ReadInt32(), + DstAlpha = (BlendFactor)reader.ReadInt32(), + AlphaOp = (BlendOp)reader.ReadInt32(), + WriteMask = (ColorComponentFlags)reader.ReadUInt32(), + }; + } + + return new PipelineKeyLogEntry + { + SettingsHash = settingsHash, + ProgramHash = new UInt128(programHigh, programLow), + Bindings = bindings, + Attributes = attributes, + ColorFormats = colors, + DepthFormat = depth, + Blend = blend, + PolygonMode = (PolygonMode)reader.ReadInt32(), + Topology = (PrimitiveTopology)reader.ReadInt32(), + }; + } +} + +/// +/// Every pipeline the renderer has used, kept next to the driver's pipeline cache so the +/// next launch can build them on a background worker before the first draw asks +/// (docs/vulkan.md#caches §6 and "Design for this renderer" item 5). +/// +/// Settings combinations make such a log grow without bound (§7), so it is capped by +/// entry count and evicts the entries whose last use is oldest. The file is versioned +/// and carries a SHA-256 of its contents; a file of another version, a truncated or +/// damaged one, reads as an empty log - it only ever costs a prewarm, never a crash. +/// +internal sealed class PipelineKeyLog +{ + /// "OPKL", little-endian. + internal const uint FileMagic = 0x4C4B504F; + + /// Bumped whenever the entry layout or the meaning of a hash in it changes. + internal const uint FormatVersion = 1; + + public const int DefaultCapacity = 4096; + + private const int HeaderSize = 4 + 4 + 4; + private const int TrailerSize = 32; + + private readonly object _lock = new(); + private readonly Dictionary _entries = new(); + private long _changes; + private long _savedChanges; + + public PipelineKeyLog(int capacity = DefaultCapacity) + { + if (capacity <= 0) throw new ArgumentOutOfRangeException(nameof(capacity)); + Capacity = capacity; + } + + public int Capacity { get; } + + public int Count + { + get { lock (_lock) return _entries.Count; } + } + + /// Something was recorded since the log was loaded or last saved. + public bool HasUnsavedChanges + { + get { lock (_lock) return _changes != _savedChanges; } + } + + /// + /// The device-wide state every pipeline of a cache is built with beyond its request: + /// the colour write tier (which states are dynamic) and dynamic blend. Program defines + /// are not here - they are part of the SPIR-V the program hash covers. + /// + public static ulong SettingsHashFor(ColorWriteTier tier, bool dynamicBlend) + { + Span hash = stackalloc byte[32]; + SHA256.HashData(Encoding.UTF8.GetBytes( + "optimum-pipeline-settings-" + FormatVersion + "|" + (int)tier + "|" + (dynamicBlend ? 1 : 0)), hash); + return BitConverter.ToUInt64(hash[..8]); + } + + /// Beside the driver cache file of the same GPU. + public static string PathFor(string cacheRoot, PipelineCacheIdentity identity) => + Path.Combine(cacheRoot, "pipeline", Path.ChangeExtension(identity.FileName, ".keys")); + + /// Notes a pipeline used now; an entry already present only moves its last-seen time. + public void Record(PipelineKeyLogEntry entry, long nowUnixMs) + { + lock (_lock) + { + if (_entries.TryGetValue(entry.ContentId, out PipelineKeyLogEntry? existing)) + { + if (existing.LastSeenUnixMs >= nowUnixMs) return; + existing.LastSeenUnixMs = nowUnixMs; + } + else + { + entry.LastSeenUnixMs = nowUnixMs; + _entries[entry.ContentId] = entry; + // Amortised: evict only once the log is a quarter over its cap. + if (_entries.Count > Capacity + Capacity / 4) TrimLocked(); + } + _changes++; + } + } + + /// The entries built for this settings hash and program. + public List Matching(ulong settingsHash, UInt128 programHash) + { + lock (_lock) + { + return _entries.Values.Where(e => e.SettingsHash == settingsHash && e.ProgramHash == programHash).ToList(); + } + } + + /// The file bytes: at most entries, most recently used first. + public byte[] Serialize() + { + lock (_lock) return SerializeLocked(); + } + + private byte[] SerializeLocked() + { + TrimLocked(); + using var stream = new MemoryStream(); + using (var writer = new BinaryWriter(stream, Encoding.UTF8, leaveOpen: true)) + { + writer.Write(FileMagic); + writer.Write(FormatVersion); + writer.Write(_entries.Count); + foreach (PipelineKeyLogEntry entry in _entries.Values.OrderByDescending(e => e.LastSeenUnixMs)) + { + entry.WriteContent(writer); + writer.Write(entry.LastSeenUnixMs); + } + } + + long length = stream.Length; + var file = new byte[length + TrailerSize]; + stream.GetBuffer().AsSpan(0, (int)length).CopyTo(file); + SHA256.HashData(file.AsSpan(0, (int)length), file.AsSpan((int)length, TrailerSize)); + return file; + } + + /// A log from file bytes; anything not written whole by this version is an empty log. + public static PipelineKeyLog Parse(byte[]? file, int capacity = DefaultCapacity) + { + var log = new PipelineKeyLog(capacity); + if (file == null || file.Length < HeaderSize + TrailerSize) return log; + if (BitConverter.ToUInt32(file, 0) != FileMagic) return log; + if (BitConverter.ToUInt32(file, 4) != FormatVersion) return log; + + int payload = file.Length - TrailerSize; + Span hash = stackalloc byte[32]; + SHA256.HashData(file.AsSpan(0, payload), hash); + if (!hash.SequenceEqual(file.AsSpan(payload, TrailerSize))) return log; + + var parsed = new Dictionary(); + try + { + using var stream = new MemoryStream(file, 0, payload, writable: false); + using var reader = new BinaryReader(stream, Encoding.UTF8); + reader.ReadUInt32(); + reader.ReadUInt32(); + int count = reader.ReadInt32(); + if (count < 0) return log; + for (int i = 0; i < count; i++) + { + PipelineKeyLogEntry? entry = PipelineKeyLogEntry.ReadContent(reader); + if (entry == null) return log; + entry.LastSeenUnixMs = reader.ReadInt64(); + parsed[entry.ContentId] = entry; + } + if (stream.Position != payload) return log; + } + catch (EndOfStreamException) + { + return log; + } + + foreach (KeyValuePair entry in parsed) log._entries.Add(entry.Key, entry.Value); + log.TrimLocked(); + return log; + } + + public static PipelineKeyLog Load(string path, int capacity = DefaultCapacity) => + Parse(CacheFileWriter.TryReadAll(path), capacity); + + /// Writes the log; false when it could not. Safe from any thread. + public bool Save(string path) + { + long changes; + byte[] bytes; + lock (_lock) + { + changes = _changes; + bytes = SerializeLocked(); + } + if (!CacheFileWriter.WriteAtomically(path, bytes)) return false; + lock (_lock) _savedChanges = Math.Max(_savedChanges, changes); + return true; + } + + /// Drops the least recently used entries beyond . + private void TrimLocked() + { + int excess = _entries.Count - Capacity; + if (excess <= 0) return; + foreach (PipelineKeyLogEntry stale in _entries.Values.OrderBy(e => e.LastSeenUnixMs).Take(excess).ToList()) + { + _entries.Remove(stale.ContentId); + } + } +} diff --git a/Optimum.Render.Vulkan/Core/PipelineState.cs b/Optimum.Render.Vulkan/Core/PipelineState.cs new file mode 100644 index 00000000..3ca7fe1c --- /dev/null +++ b/Optimum.Render.Vulkan/Core/PipelineState.cs @@ -0,0 +1,405 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using System.Numerics; + +namespace Optimum.Render.Vulkan.Core; + +/// Blend configuration for one colour attachment. +internal struct AttachmentBlend : IEquatable +{ + public bool Enabled; + public BlendFactor SrcColor; + public BlendFactor DstColor; + public BlendOp ColorOp; + public BlendFactor SrcAlpha; + public BlendFactor DstAlpha; + public BlendOp AlphaOp; + public ColorComponentFlags WriteMask; + + public static AttachmentBlend Default => new() + { + Enabled = false, + SrcColor = BlendFactor.SrcAlpha, + DstColor = BlendFactor.OneMinusSrcAlpha, + ColorOp = BlendOp.Add, + SrcAlpha = BlendFactor.SrcAlpha, + DstAlpha = BlendFactor.OneMinusSrcAlpha, + AlphaOp = BlendOp.Add, + WriteMask = ColorComponentFlags.RBit | ColorComponentFlags.GBit + | ColorComponentFlags.BBit | ColorComponentFlags.ABit, + }; + + /// + /// The factor pairs one of the game's named blend modes means, which + /// ClientPlatformWindows.GlToggleBlend selects and + /// the platform's stated state (StatedRenderState) applies to every attachment. + /// + /// A native render system states its blend outright rather than reading the tracker's + /// (docs/vulkan.md, decision 3), and its call site says "blend on, + /// standard" the same way the OpenGL body does, so it builds the attachment through here + /// instead of restating the factors and risking a pair that drifts from the table. + /// + public static AttachmentBlend For(bool enabled, EnumBlendMode mode) + { + (BlendFactor srcColor, BlendFactor dstColor, BlendFactor srcAlpha, BlendFactor dstAlpha) = FactorsFor(mode); + AttachmentBlend blend = Default; + blend.Enabled = enabled; + blend.SrcColor = srcColor; + blend.DstColor = dstColor; + blend.ColorOp = BlendOp.Add; + blend.SrcAlpha = srcAlpha; + blend.DstAlpha = dstAlpha; + blend.AlphaOp = BlendOp.Add; + return blend; + } + + /// + /// World/UI separation: the blend a draw into the UI image uses in place of this one. + /// + /// gui.fsh writes straight alpha and the GUI draws under , + /// whose factors are not separate - (SRC_ALPHA, ONE_MINUS_SRC_ALPHA) on the alpha channel too. + /// Onto the opaque window that is correct, which is why it always was; accumulated into an image + /// that starts transparent it gives out_a = src_a * src_a + dst_a * (1 - src_a), roughly alpha + /// squared per layer, instead of the over-operator's out_a = src_a + dst_a * (1 - src_a), and the + /// compose then shows the world through every translucent panel. So Standard's exact factor set + /// takes ONE for the source alpha factor here, and nothing else changes: the RGB factors already + /// accumulate the premultiplied colour the compose blends back with (ONE, ONE_MINUS_SRC_ALPHA). + /// + /// Only Standard, deliberately. PremultipliedAlpha already has (ONE, ONE_MINUS_SRC_ALPHA) on both + /// channels, and the destination-reading modes (Brighten, Multiply, Glow, Overlay) belong to world + /// systems that draw before the blit, never into the UI image. Pinned by UiSeparationTests. + /// + public AttachmentBlend ForUiImage() + { + AttachmentBlend blend = this; + if (blend.SrcColor == BlendFactor.SrcAlpha && blend.DstColor == BlendFactor.OneMinusSrcAlpha && + blend.SrcAlpha == BlendFactor.SrcAlpha && blend.DstAlpha == BlendFactor.OneMinusSrcAlpha) + { + blend.SrcAlpha = BlendFactor.One; + } + return blend; + } + + /// The one table of factor pairs, shared by the stated state and by native systems. + internal static (BlendFactor SrcColor, BlendFactor DstColor, BlendFactor SrcAlpha, BlendFactor DstAlpha) + FactorsFor(EnumBlendMode mode) => mode switch + { + EnumBlendMode.Brighten => (BlendFactor.DstColor, BlendFactor.One, + BlendFactor.DstColor, BlendFactor.One), + EnumBlendMode.Multiply => (BlendFactor.Zero, BlendFactor.OneMinusSrcAlpha, + BlendFactor.One, BlendFactor.OneMinusSrcAlpha), + EnumBlendMode.PremultipliedAlpha => (BlendFactor.One, BlendFactor.OneMinusSrcAlpha, + BlendFactor.One, BlendFactor.OneMinusSrcAlpha), + EnumBlendMode.Glow => (BlendFactor.SrcAlpha, BlendFactor.One, + BlendFactor.One, BlendFactor.Zero), + EnumBlendMode.Overlay => (BlendFactor.SrcAlpha, BlendFactor.OneMinusSrcAlpha, + BlendFactor.One, BlendFactor.One), + _ => (BlendFactor.SrcAlpha, BlendFactor.OneMinusSrcAlpha, + BlendFactor.SrcAlpha, BlendFactor.OneMinusSrcAlpha), + }; + + /// + /// Squeezes the whole attachment state into 32 bits so a set of eight hashes + /// as cheaply as an array of ints. Every field is a small enum; the widest is + /// a blend factor at 19 values. + /// + public readonly uint Pack() + { + uint packed = Enabled ? 1u : 0u; + packed |= (uint)SrcColor << 1; + packed |= (uint)DstColor << 6; + packed |= (uint)ColorOp << 11; + packed |= (uint)SrcAlpha << 14; + packed |= (uint)DstAlpha << 19; + packed |= (uint)AlphaOp << 24; + packed |= (uint)WriteMask << 27; + return packed; + } + + public readonly bool Equals(AttachmentBlend other) => Pack() == other.Pack(); + public override readonly bool Equals(object? obj) => obj is AttachmentBlend other && Equals(other); + public override readonly int GetHashCode() => (int)Pack(); +} + +/// +/// Interns a value so it can be compared as an int. +/// +/// The pipeline key is looked up on every draw, so it has to be small and cheap +/// to hash. Interning the bulky parts - the blend set, the render target formats, +/// the vertex layout - turns each into one integer and leaves the key at six. +/// +internal sealed class Interner where T : notnull +{ + private readonly Dictionary _ids; + private readonly List _values = new(); + + public Interner(IEqualityComparer? comparer = null) => _ids = new Dictionary(comparer); + + public int Intern(T value) + { + if (_ids.TryGetValue(value, out int id)) return id; + + id = _values.Count; + _values.Add(value); + _ids[value] = id; + return id; + } + + public T Get(int id) => _values[id]; + public int Count => _values.Count; +} + +/// The attachment formats a pipeline renders into. +internal sealed class RenderTargetFormats : IEquatable +{ + public Format[] ColorFormats { get; } + public Format DepthFormat { get; } + + public RenderTargetFormats(Format[] colorFormats, Format depthFormat) + { + ColorFormats = colorFormats; + DepthFormat = depthFormat; + } + + public bool Equals(RenderTargetFormats? other) + { + if (other is null) return false; + if (DepthFormat != other.DepthFormat) return false; + if (ColorFormats.Length != other.ColorFormats.Length) return false; + + for (int i = 0; i < ColorFormats.Length; i++) + { + if (ColorFormats[i] != other.ColorFormats[i]) return false; + } + return true; + } + + public override bool Equals(object? obj) => Equals(obj as RenderTargetFormats); + + public override int GetHashCode() + { + var hash = new HashCode(); + hash.Add(DepthFormat); + foreach (Format format in ColorFormats) hash.Add(format); + return hash.ToHashCode(); + } +} + +/// A set of per-attachment blend states, interned as a unit without heap storage. +internal readonly struct BlendSignature : IEquatable +{ + private readonly UInt128 _first; + private readonly UInt128 _second; + private readonly byte _count; + + public BlendSignature(ReadOnlySpan attachments) : this(attachments, attachments.Length) { } + + public BlendSignature(ReadOnlySpan attachments, int count) + { + if ((uint)count > RenderLimits.MaxColorAttachments) + throw new ArgumentOutOfRangeException(nameof(count)); + + _count = (byte)count; + UInt128 first = 0; + UInt128 second = 0; + for (int i = 0; i < count; i++) + { + uint packed = (i < attachments.Length ? attachments[i] : AttachmentBlend.Default).Pack(); + if (i < 4) first |= (UInt128)packed << (i * 32); + else second |= (UInt128)packed << ((i - 4) * 32); + } + _first = first; + _second = second; + } + + public bool Equals(BlendSignature other) => + _count == other._count && _first == other._first && _second == other._second; + + public override bool Equals(object? obj) => obj is BlendSignature other && Equals(other); + public override int GetHashCode() => HashCode.Combine(_count, _first, _second); +} + +/// +/// Everything a graphics pipeline is built from that Vulkan cannot change +/// dynamically. +/// +/// Vulkan 1.3 makes viewport, scissor, cull mode, front face, depth test/write/ +/// compare, stencil state and line width dynamic, so none of them appear here and +/// none of them cause a pipeline to be created. What is left is the shader +/// program, the vertex layout, the attachment formats, the blend set, the fill +/// mode and the topology class - and all but the last two are interned to an int. +/// +internal readonly record struct PipelineKey( + int ProgramId, + int VertexLayoutId, + int TargetFormatsId, + int BlendId, + PolygonMode PolygonMode, + int TopologyClass); + +/// +/// The fixed limits every target and program of this renderer is built within, and the one +/// winding the game uses. +/// +internal static class RenderLimits +{ + public const int MaxColorAttachments = 8; + public const int MaxTextureUnits = 16; + + /// + /// The front face is a constant, not a setting. GL's counter-clockwise + /// winding, read in a Vulkan framebuffer with no Y flip, is clockwise. The + /// game never calls glFrontFace, so nothing varies it. + /// + public const FrontFace FrontFace = Silk.NET.Vulkan.FrontFace.Clockwise; + + /// Bit i set when the program statically writes fragment output i. + public static uint OutputBits(HashSet writtenOutputs) + { + uint bits = 0; + for (int i = 0; i < MaxColorAttachments; i++) + { + if (writtenOutputs.Contains(i)) bits |= 1u << i; + } + return bits; + } +} + +/// One bit per dynamic-state command a draw may record. +[Flags] +internal enum DynamicStateDirty : ushort +{ + None = 0, + Viewport = 1 << 0, + Scissor = 1 << 1, + CullMode = 1 << 2, + FrontFace = 1 << 3, + Topology = 1 << 4, + DepthTestEnable = 1 << 5, + DepthWriteEnable = 1 << 6, + DepthCompareOp = 1 << 7, + StencilTestEnable = 1 << 8, + StencilOp = 1 << 9, + StencilCompareMask = 1 << 10, + StencilWriteMask = 1 << 11, + StencilReference = 1 << 12, + LineWidth = 1 << 13, + /// The Vulkan 1.3 core set every pipeline declares dynamic. + All = (1 << 14) - 1, + /// vkCmdSetColorWriteEnableEXT or vkCmdSetColorWriteMaskEXT, per the colour write tier. + ColorWrite = 1 << 14, + /// vkCmdSetColorBlendEnableEXT + vkCmdSetColorBlendEquationEXT (mask tier with dynamic blend). + ColorBlend = 1 << 15, + /// What a fresh recording marks dirty; the device drops the bits its tier does not use. + Everything = All | ColorWrite | ColorBlend, +} + +/// The values a draw's dynamic state resolves to, already in Vulkan terms. +internal struct DynamicStateValues +{ + public Viewport Viewport; + public Rect2D Scissor; + public CullModeFlags CullMode; + public FrontFace FrontFace; + public PrimitiveTopology Topology; + public bool DepthTest; + public bool DepthWrite; + public CompareOp DepthCompare; + public bool StencilTest; + public StencilOp StencilFail; + public StencilOp StencilPass; + public StencilOp StencilDepthFail; + public CompareOp StencilCompare; + public uint StencilCompareMask; + public uint StencilWriteMask; + public uint StencilReference; + public float LineWidth; + /// + /// The colour write state the tier makes dynamic: enable bits (enable tier) or + /// the effective masks packed four bits per attachment (mask tier); 0 otherwise. + /// + public uint ColorWrite; + /// Interned id of the full per-attachment blend set (mask tier with dynamic blend); 0 otherwise. + public int BlendStateId; +} + +/// +/// What the command buffer being recorded already holds, so a draw emits only +/// the dynamic state that changed. +/// +/// Dynamic state is command-buffer state: it survives rendering scopes and +/// pipeline binds (every pipeline declares all of these dynamic), and is +/// undefined again when a command buffer begins. The cache is keyed on the +/// recording serial of the command buffer (), +/// so a new frame, a partial submission's continuation or a recycled handle +/// always starts from "everything dirty". +/// +internal sealed class DynamicStateCache +{ + private ulong _serial; + private DynamicStateValues _last; + + /// False emits everything on every draw, as before masking. Tests compare the two. + public bool Enabled { get; set; } = true; + + /// Forgets what was recorded; the next draw emits everything. + public void Invalidate() => _serial = 0; + + /// + /// Returns the commands a draw recorded into the command buffer with + /// has to emit for , and + /// assumes the caller emits them. A serial of 0 is never trusted. + /// + public DynamicStateDirty Update(ulong serial, in DynamicStateValues next) + { + DynamicStateDirty dirty; + if (!Enabled || serial == 0 || serial != _serial) + { + dirty = DynamicStateDirty.Everything; + } + else + { + dirty = DynamicStateDirty.None; + if (!SameViewport(_last.Viewport, next.Viewport)) dirty |= DynamicStateDirty.Viewport; + if (!SameRect(_last.Scissor, next.Scissor)) dirty |= DynamicStateDirty.Scissor; + if (_last.CullMode != next.CullMode) dirty |= DynamicStateDirty.CullMode; + if (_last.FrontFace != next.FrontFace) dirty |= DynamicStateDirty.FrontFace; + if (_last.Topology != next.Topology) dirty |= DynamicStateDirty.Topology; + if (_last.DepthTest != next.DepthTest) dirty |= DynamicStateDirty.DepthTestEnable; + if (_last.DepthWrite != next.DepthWrite) dirty |= DynamicStateDirty.DepthWriteEnable; + if (_last.DepthCompare != next.DepthCompare) dirty |= DynamicStateDirty.DepthCompareOp; + if (_last.StencilTest != next.StencilTest) dirty |= DynamicStateDirty.StencilTestEnable; + if (_last.StencilFail != next.StencilFail || _last.StencilPass != next.StencilPass || + _last.StencilDepthFail != next.StencilDepthFail || _last.StencilCompare != next.StencilCompare) + { + dirty |= DynamicStateDirty.StencilOp; + } + if (_last.StencilCompareMask != next.StencilCompareMask) dirty |= DynamicStateDirty.StencilCompareMask; + if (_last.StencilWriteMask != next.StencilWriteMask) dirty |= DynamicStateDirty.StencilWriteMask; + if (_last.StencilReference != next.StencilReference) dirty |= DynamicStateDirty.StencilReference; + // Bitwise, so a NaN width is not "changed" forever. + if (BitConverter.SingleToInt32Bits(_last.LineWidth) != BitConverter.SingleToInt32Bits(next.LineWidth)) + { + dirty |= DynamicStateDirty.LineWidth; + } + if (_last.ColorWrite != next.ColorWrite) dirty |= DynamicStateDirty.ColorWrite; + if (_last.BlendStateId != next.BlendStateId) dirty |= DynamicStateDirty.ColorBlend; + } + + _serial = serial; + _last = next; + return dirty; + } + + public static int CommandCount(DynamicStateDirty dirty) => BitOperations.PopCount((uint)dirty); + + private static bool SameViewport(in Viewport a, in Viewport b) => + a.X == b.X && a.Y == b.Y && a.Width == b.Width && a.Height == b.Height && + a.MinDepth == b.MinDepth && a.MaxDepth == b.MaxDepth; + + private static bool SameRect(in Rect2D a, in Rect2D b) => + a.Offset.X == b.Offset.X && a.Offset.Y == b.Offset.Y && + a.Extent.Width == b.Extent.Width && a.Extent.Height == b.Extent.Height; +} diff --git a/Optimum.Render.Vulkan/Core/RenderTargetManager.cs b/Optimum.Render.Vulkan/Core/RenderTargetManager.cs new file mode 100644 index 00000000..bfa43036 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/RenderTargetManager.cs @@ -0,0 +1,790 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// One attachment slot of a framebuffer. +internal struct AttachmentSlot +{ + public int TextureId; + public uint Layer; + + public readonly bool IsBound => TextureId > 0; +} + +/// +/// A render target: colour attachments in GL's positional slots, an optional +/// depth attachment, and the draw-buffer mask. +/// +internal sealed class VulkanFramebuffer +{ + public int Id; + public uint Width; + public uint Height; + + public AttachmentSlot[] Color = new AttachmentSlot[RenderLimits.MaxColorAttachments]; + public int DepthTextureId; + + /// Cached interned id of the attachment formats, or -1 when stale. + public int FormatsId = -1; + + /// + /// Bound colour slots the declared frame-graph pass leaves out of its scope (the + /// final composition writes Primary 0 and samples Primary 1). Cleared when the pass ends. + /// + public uint PassExclusion; +} + +/// +/// Owns framebuffers and drives dynamic rendering scopes. +/// +/// The subtle part is glDrawBuffers. Since Phase 2 (contract C4) it is a +/// write mask, never a scope restart: the scope carries every bound colour slot, +/// and a cleared draw-buffer bit only zeroes that attachment's effective write +/// mask (dynamic enable, dynamic mask or pipeline key, per +/// ). The TAA motion windows toggle the motion +/// attachment that way inside one scope. +/// +/// The exception is a read: the final composition pass renders into the primary +/// framebuffer's attachment 0 while sampling its attachment 1, which Vulkan only +/// allows with attachment 1 out of the scope. A draw that samples a bound slot +/// whose draw buffer is off therefore leaves that slot out (a null attachment, +/// keeping fragment output N aimed at slot N) until its draw buffer is selected +/// again or the framebuffer is rebound; those restarts are feedback splits. +/// +internal sealed unsafe class RenderTargetManager : IDisposable +{ + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly Interner _formats = new(); + + /// Every attachment of a scope moves in one barrier command before vkCmdBeginRendering. + private readonly BarrierBatcher _barriers; + + private readonly List _framebuffers = new(); + private readonly Stack _freeIds = new(); + + private VulkanFramebuffer? _bound; + private bool _renderingActive; + private bool _disposed; + + /// How many rendering scopes have been opened, for diagnostics. + public long ScopesOpened { get; private set; } + + /// + /// Restarts that reopened exactly the attachment set they closed (views and + /// layouts). Draw-buffer and colour-mask changes never restart, so this stays 0. + /// + public long MaskRestarts { get; private set; } + + /// Restarts that left a sampled, draw-buffer-excluded slot out of the scope or let it rejoin. + public long FeedbackSplits { get; private set; } + + // What the open scope was begun with, to recognise a restart that changed nothing. + // Scratch is used only while preparing a scope; pass signatures own their snapshots. + private readonly VulkanTexture?[] _scopeColour = new VulkanTexture?[RenderLimits.MaxColorAttachments]; + private readonly ImageView[] _openViews = new ImageView[RenderLimits.MaxColorAttachments]; + private int _openCount = -1; + private ImageView _openDepthView; + private ImageLayout _openDepthLayout; + + /// + /// Runs right after vkCmdBeginRendering, inside the new scope. The + /// occlusion query ring resumes a query the previous scope's end suspended: + /// GL counts samples across framebuffer changes, Vulkan only within a scope. + /// + public Action? ScopeOpened; + + /// Runs right before vkCmdEndRendering, still inside the scope (ends a running query). + public Action? ScopeClosing; + + /// Runs right after vkCmdEndRendering, outside any scope (where a query pool may be reset). + public Action? ScopeClosed; + + /// The frame graph: declared passes, the plan and promoted clears (Phase 2 step 2). + private readonly FrameGraph _graph; + + /// Opens the one scope of each pass on the frame-graph path. + private readonly PassRecorder _recorder; + + public RenderTargetManager(VulkanContext context, TextureManager textures, FrameGraph? graph = null) + { + _context = context; + _textures = textures; + _barriers = textures.CreateBatcher(); + _graph = graph ?? new FrameGraph { Enabled = false }; + _recorder = new PassRecorder(context, textures, _barriers, _graph); + + // Index 0 is the default framebuffer, installed separately. + _framebuffers.Add(null); + } + + public VulkanFramebuffer? Bound => _bound; + + public FrameGraph Graph => _graph; + + /// The declared pass, while one is current (frame-graph path only). + public PassDeclaration? DeclaredPass => _recorder.Declared; + public bool RenderingActive => _renderingActive; + + public VulkanFramebuffer? Get(int id) => + id > 0 && id < _framebuffers.Count ? _framebuffers[id] : null; + + public int Create(uint width, uint height) + { + var framebuffer = new VulkanFramebuffer { Width = width, Height = height }; + + if (_freeIds.Count > 0) + { + int reused = _freeIds.Pop(); + framebuffer.Id = reused; + _framebuffers[reused] = framebuffer; + return reused; + } + + _framebuffers.Add(framebuffer); + framebuffer.Id = _framebuffers.Count - 1; + return framebuffer.Id; + } + + public void Attach(int framebufferId, int attachmentIndex, int textureId, uint layer = 0) + { + VulkanFramebuffer? framebuffer = Get(framebufferId); + if (framebuffer == null) return; + + if (attachmentIndex < 0) + { + framebuffer.DepthTextureId = textureId; + } + else if (attachmentIndex < RenderLimits.MaxColorAttachments) + { + AttachmentSlot previous = framebuffer.Color[attachmentIndex]; + if (previous.TextureId == textureId && previous.Layer == layer) return; + framebuffer.Color[attachmentIndex] = new AttachmentSlot { TextureId = textureId, Layer = layer }; + } + + framebuffer.FormatsId = -1; + + // GL attaches to the bound framebuffer, so an open scope no longer + // describes the target: the next draw must reopen on the new views. + if (_bound == framebuffer) _needsRestart = true; + } + + private void NoteFeedbackSplit() + { + FeedbackSplits++; + VulkanStats.NoteFeedbackSplit(); + } + + /// Whether colour slot is part of the scope the framebuffer opens. + private static bool InScope(VulkanFramebuffer framebuffer, int index) => + framebuffer.Color[index].IsBound && ((framebuffer.PassExclusion >> index) & 1) == 0; + + private bool _needsRestart; + + /// + /// Whether the scope holds its depth attachment in the read-only layout. + /// + /// GL lets a pass sample the depth buffer it is drawing against as long as + /// depth writes are off - the liquid pass reads scene depth that way to fade + /// water at its edges. Vulkan allows the same only if the attachment is in + /// DEPTH_READ_ONLY_OPTIMAL for both the attachment and the descriptor, so + /// the scope switches layout for such draws and back for the next that + /// writes depth. + /// + public bool DepthReadOnly { get; private set; } + + public void SetDepthReadOnly(bool readOnly) + { + if (DepthReadOnly == readOnly) return; + DepthReadOnly = readOnly; + if (_renderingActive) _needsRestart = true; + } + + /// Whether a texture is the bound framebuffer's depth attachment. + public bool IsBoundDepth(int textureId) => + _bound != null && textureId > 0 && _bound.DepthTextureId == textureId; + + /// + /// Whether a texture takes part in the rendering scope the bound framebuffer + /// is about to open, and so has to keep its attachment layout. + /// + /// Only slots the draw actually writes count. A colour attachment masked out + /// of glDrawBuffers is not part of the scope at all, and the composition pass + /// samples exactly such a slot - so it has to stay transitionable, or it is + /// read in the colour-attachment layout it was left in. + /// + public bool IsAttachmentOfBound(int textureId) + { + if (_bound == null || textureId <= 0) return false; + if (_bound.DepthTextureId == textureId) return true; + + for (int i = 0; i < _bound.Color.Length; i++) + { + if (_bound.Color[i].TextureId != textureId) continue; + // Left out of the declared pass's scope: sampled directly, not feedback. + if (((_bound.PassExclusion >> i) & 1) != 0) continue; + return true; + } + return false; + } + + /// + /// Binds a framebuffer. Nothing is recorded here: GL lets a bind be followed + /// by more state changes before anything is drawn, so the scope opens lazily + /// at the first draw or clear. + /// + public void Bind(CommandBuffer commandBuffer, int framebufferId) + { + VulkanFramebuffer? framebuffer = Get(framebufferId); + if (ReferenceEquals(framebuffer, _bound)) return; + + EndRendering(commandBuffer); + // A pass is declared on one target; binding another ends it. + if (_recorder.Declared != null) ClearPassDeclaration(); + _bound = framebuffer; + _needsRestart = false; + } + + public void Delete(int framebufferId) + { + VulkanFramebuffer? framebuffer = Get(framebufferId); + if (framebuffer == null) return; + + if (ReferenceEquals(framebuffer, _bound)) _bound = null; + if (ReferenceEquals(framebuffer, _recorder.DeclaredOn)) ClearPassDeclaration(); + _framebuffers[framebufferId] = null; + _freeIds.Push(framebufferId); + } + + // ------------------------------------------------------------------- passes + + /// + /// Declares a frame-graph pass on (0: the bound + /// target), binding it. The pass's scope opens lazily at its first draw or in-pass + /// clear, exactly once unless something forces a split. Re-declaring the current + /// pass (same name, target and slots) changes nothing. With the frame graph off this + /// only binds, so the same frame code drives both paths. + /// + public void DeclarePass(CommandBuffer commandBuffer, PassDeclaration declaration, int framebufferId) + { + if (!_graph.Enabled) + { + // No pass bookkeeping, but the scope still holds only the declared slots. + if (framebufferId <= 0) return; + Bind(commandBuffer, framebufferId); + ApplyPassExclusion(commandBuffer, _bound!, declaration.ColorSlots); + return; + } + + VulkanFramebuffer? target = framebufferId > 0 ? Get(framebufferId) : _bound; + if (target == null) + { + EndPass(commandBuffer); + return; + } + + PassDeclaration? current = _recorder.Declared; + if (current != null && ReferenceEquals(_recorder.DeclaredOn, target) && ReferenceEquals(_bound, target) && + current.Name == declaration.Name && current.ColorSlots == declaration.ColorSlots) + { + return; + } + + EndPass(commandBuffer); + Bind(commandBuffer, target.Id); + + ApplyPassExclusion(commandBuffer, target, declaration.ColorSlots); + _recorder.Declare(declaration, target); + } + + /// + /// A new pass is a new use of the target: every bound slot is in its scope except the slots + /// the pass leaves out, which can then be sampled. A change reopens an open scope. + /// + private void ApplyPassExclusion(CommandBuffer commandBuffer, VulkanFramebuffer target, uint colorSlots) + { + uint exclusion = 0; + for (int i = 0; i < target.Color.Length; i++) + { + if (target.Color[i].IsBound && ((colorSlots >> i) & 1) == 0) exclusion |= 1u << i; + } + if (target.PassExclusion == exclusion) return; + target.PassExclusion = exclusion; + target.FormatsId = -1; + if (ReferenceEquals(target, _bound) && _renderingActive) EndRendering(commandBuffer); + } + + /// Ends the current pass, declared or not: closes its scope. No-op with the frame graph off. + public void EndPass(CommandBuffer commandBuffer) + { + if (!_graph.Enabled) return; + EndRendering(commandBuffer); + ClearPassDeclaration(); + } + + private void ClearPassDeclaration() + { + VulkanFramebuffer? target = _recorder.DeclaredOn; + if (target != null && target.PassExclusion != 0) + { + target.PassExclusion = 0; + target.FormatsId = -1; + if (ReferenceEquals(target, _bound) && _renderingActive) _needsRestart = true; + } + _recorder.ClearDeclaration(); + } + + /// + /// Records the clears promoted into as clear-image commands, + /// closing an open scope first: the texture is about to be used some other way (sampled, + /// copied, read back, uploaded to) before any pass attached it. + /// + public void FlushPendingClears(CommandBuffer commandBuffer, VulkanTexture texture) + { + if (!_graph.HasPendingClears || !_graph.HasPendingClear(texture)) return; + EndRendering(commandBuffer); + _recorder.FlushClears(commandBuffer, texture); + } + + /// Every clear still pending at the end of the frame lands as a clear-image command. + public void FlushAllPendingClears(CommandBuffer commandBuffer) + { + if (!_graph.HasPendingClears) return; + EndRendering(commandBuffer); + _recorder.FlushClears(commandBuffer, null); + } + + /// A deleted texture's pending clears are dropped. + public void DropPendingClears(VulkanTexture texture) => _graph.Drop(texture); + + // ------------------------------------------------------------------- scopes + + /// + /// Opens a rendering scope if one is not already open, transitioning every + /// participating attachment into its attachment layout. Every bound colour + /// slot participates unless the declared pass leaves it out; a pass that samples + /// such a slot moves it to a shader-readable layout itself. + /// + public void EnsureRendering(CommandBuffer commandBuffer) + { + if (_renderingActive && !_needsRestart) return; + if (_bound == null) return; + + bool restarting = _renderingActive; + if (_renderingActive) EndRendering(commandBuffer); + + VulkanFramebuffer framebuffer = _bound; + int highest = HighestScopeAttachment(framebuffer); + int count = highest + 1; + bool graph = _graph.Enabled; + + Span attachments = stackalloc RenderingAttachmentInfo[count]; + // Frame-graph path: the pass recorder queues the barriers and picks the load ops. + Span scopeColour = _scopeColour.AsSpan(0, count); + scopeColour.Clear(); + VulkanTexture? scopeDepth = null; + + for (int i = 0; i < count; i++) + { + AttachmentSlot slot = framebuffer.Color[i]; + + if (!InScope(framebuffer, i)) + { + // A null view keeps fragment output i pointed at slot i: an + // unbound slot, or one the declared pass leaves out. + attachments[i] = new RenderingAttachmentInfo + { + SType = StructureType.RenderingAttachmentInfo, + ImageView = default, + ImageLayout = ImageLayout.Undefined, + LoadOp = AttachmentLoadOp.DontCare, + StoreOp = AttachmentStoreOp.DontCare, + }; + continue; + } + + VulkanTexture? texture = _textures.Get(slot.TextureId); + if (texture == null) + { + attachments[i] = new RenderingAttachmentInfo { SType = StructureType.RenderingAttachmentInfo }; + continue; + } + + // Blend state can change inside the scope, so the attachment is + // declared for the widest colour use (read and write). + if (graph) scopeColour[i] = texture; + else _textures.Require(_barriers, commandBuffer, texture, ResourceUsage.ColorBlend); + + attachments[i] = new RenderingAttachmentInfo + { + SType = StructureType.RenderingAttachmentInfo, + // The slot's layer, not the whole image: an array attached once + // per layer must reach a different layer each time. + ImageView = texture.ViewOfLayer(slot.Layer), + ImageLayout = ImageLayout.ColorAttachmentOptimal, + // LOAD preserves what is already there, which is GL's model: a + // framebuffer keeps its contents until something clears it. + LoadOp = AttachmentLoadOp.Load, + StoreOp = AttachmentStoreOp.Store, + }; + } + + RenderingAttachmentInfo depthAttachment = default; + bool hasDepth = false; + if (framebuffer.DepthTextureId > 0) + { + VulkanTexture? depth = _textures.Get(framebuffer.DepthTextureId); + if (depth != null) + { + ImageLayout depthLayout = DepthReadOnly + ? ImageLayout.DepthReadOnlyOptimal + : ImageLayout.DepthAttachmentOptimal; + // Read-only depth may be sampled by the draws of this scope. + if (graph) scopeDepth = depth; + else _textures.Require(_barriers, commandBuffer, depth, + DepthReadOnly ? ResourceUsage.DepthReadOnlySampled : ResourceUsage.DepthWrite); + depthAttachment = new RenderingAttachmentInfo + { + SType = StructureType.RenderingAttachmentInfo, + ImageView = depth.View, + ImageLayout = depthLayout, + LoadOp = AttachmentLoadOp.Load, + StoreOp = AttachmentStoreOp.Store, + }; + hasDepth = true; + } + } + + if (graph) + { + _recorder.Prepare(commandBuffer, framebuffer, scopeColour, scopeDepth, DepthReadOnly, + FormatsIdOf(framebuffer), _framebuffers, attachments, ref depthAttachment); + scopeColour.Clear(); + } + + _barriers.Flush(commandBuffer); + + fixed (RenderingAttachmentInfo* attachmentsPtr = attachments) + { + var rendering = new RenderingInfo + { + SType = StructureType.RenderingInfo, + RenderArea = new Rect2D(new Offset2D(0, 0), new Extent2D(framebuffer.Width, framebuffer.Height)), + LayerCount = 1, + ColorAttachmentCount = (uint)attachments.Length, + PColorAttachments = attachments.Length == 0 ? null : attachmentsPtr, + PDepthAttachment = hasDepth ? &depthAttachment : null, + }; + + _context.Api.CmdBeginRendering(commandBuffer, &rendering); + } + + // A restart that reopened the very set it closed bought nothing: with + // write masks carrying draw buffers and motion windows this never happens. + ImageView depthView = hasDepth ? depthAttachment.ImageView : default; + ImageLayout depthLayoutOpened = hasDepth ? depthAttachment.ImageLayout : ImageLayout.Undefined; + bool unchanged = restarting && count == _openCount && depthView.Handle == _openDepthView.Handle + && depthLayoutOpened == _openDepthLayout; + for (int i = 0; unchanged && i < count; i++) + { + unchanged = attachments[i].ImageView.Handle == _openViews[i].Handle; + } + if (unchanged) + { + MaskRestarts++; + VulkanStats.NoteMaskRestart(); + } + _openCount = count; + _openDepthView = depthView; + _openDepthLayout = depthLayoutOpened; + for (int i = 0; i < count; i++) _openViews[i] = attachments[i].ImageView; + + _renderingActive = true; + _needsRestart = false; + ScopesOpened++; + VulkanStats.NoteScopeOpened(); + ScopeOpened?.Invoke(commandBuffer); + } + + /// + /// Closes the scope. Uploads and layout transitions have to happen outside + /// one, so this is called before them and the scope reopens on the next draw. + /// + public void EndRendering(CommandBuffer commandBuffer) + { + if (!_renderingActive) return; + ScopeClosing?.Invoke(commandBuffer); + _context.Api.CmdEndRendering(commandBuffer); + _renderingActive = false; + ScopeClosed?.Invoke(commandBuffer); + } + + // ------------------------------------------------------------------- clears + + /// + /// Clears one colour slot of a declared native pass (docs/vulkan.md, + /// decision 4). Unlike this consults no draw-buffer mask and no + /// tracked colour mask - a native pass states the slots it writes, and a slot it states is a + /// slot it may clear. With the frame graph on and no scope open yet the clear is promoted, so + /// it becomes the scope's load op instead of a second write. + /// + public void ClearPassAttachment(CommandBuffer commandBuffer, int attachment, float r, float g, float b, float a) + { + if (_bound == null) return; + if ((uint)attachment >= (uint)_bound.Color.Length) return; + if (!_bound.Color[attachment].IsBound) return; + + if (_graph.Enabled && (!_renderingActive || _needsRestart)) + { + VulkanTexture? texture = _textures.Get(_bound.Color[attachment].TextureId); + if (texture == null) return; + EndRendering(commandBuffer); + _graph.PromoteColorClear(texture, _bound.Color[attachment].Layer, r, g, b, a); + return; + } + + EnsureRendering(commandBuffer); + if (!_renderingActive) return; + if (_graph.Enabled) _graph.NoteInPassClear(); + + var clear = new ClearAttachment + { + AspectMask = ImageAspectFlags.ColorBit, + ColorAttachment = (uint)attachment, + ClearValue = new ClearValue(new ClearColorValue(r, g, b, a)), + }; + var rect = new ClearRect + { + Rect = new Rect2D(new Offset2D(0, 0), new Extent2D(_bound.Width, _bound.Height)), + BaseArrayLayer = 0, + LayerCount = 1, + }; + _context.Api.CmdClearAttachments(commandBuffer, 1, &clear, 1, &rect); + } + + /// + /// A colour clear of the bound target. The caller has applied the draw buffers and colour mask + /// the client stated (VulkanClientPlatform.ClearTargetColor): glClearBuffer on a draw buffer + /// glDrawBuffers left out, or through an all-false glColorMask, never reaches here. + /// + public void ClearColor(CommandBuffer commandBuffer, int attachment, float r, float g, float b, float a) + { + if (_bound == null) return; + + // glClearBuffer names a draw buffer, and one that glDrawBuffers left out + // is simply not cleared - the game clears attachments 2 and 3 of the + // primary target while only 0 and 1 are selected. Nor does GL clear + // through an all-false glColorMask. A clear on an attachment whose + // effective write mask is zero is a no-op on every path: GL keeps an attachment the + // shader never writes, and Vulkan would write garbage into it. Both rules are the + // platform's now, applied to the draw buffers and mask it stated before calling here. + if ((uint)attachment >= (uint)_bound.Color.Length) return; + if (!_bound.Color[attachment].IsBound) return; + + if (_graph.Enabled && !ClearColorOnGraph(commandBuffer, attachment, r, g, b, a)) return; + if (!_graph.Enabled && !InScope(_bound, attachment)) + { + // Left out of the declared pass: GL still clears the texture, outside any scope. + ClearImage(commandBuffer, _bound.Color[attachment], r, g, b, a); + return; + } + if (!_graph.Enabled) EnsureRendering(commandBuffer); + if (!_renderingActive) return; + + var clear = new ClearAttachment + { + AspectMask = ImageAspectFlags.ColorBit, + ColorAttachment = (uint)attachment, + ClearValue = new ClearValue(new ClearColorValue(r, g, b, a)), + }; + var rect = new ClearRect + { + Rect = new Rect2D(new Offset2D(0, 0), new Extent2D(_bound.Width, _bound.Height)), + BaseArrayLayer = 0, + LayerCount = 1, + }; + _context.Api.CmdClearAttachments(commandBuffer, 1, &clear, 1, &rect); + } + + private void ClearImage(CommandBuffer commandBuffer, AttachmentSlot slot, float r, float g, float b, float a) + { + VulkanTexture? texture = _textures.Get(slot.TextureId); + if (texture == null) return; + EndRendering(commandBuffer); + _textures.Require(_barriers, commandBuffer, texture, ResourceUsage.TransferDst); + _barriers.Flush(commandBuffer); + var value = new ClearColorValue(r, g, b, a); + var range = new ImageSubresourceRange(ImageAspectFlags.ColorBit, 0, 1, slot.Layer, 1); + _context.Api.CmdClearColorImage(commandBuffer, texture.Image, ImageLayout.TransferDstOptimal, &value, 1, &range); + } + + /// + /// The frame-graph half of a colour clear. A slot left out by the declared pass is not in + /// the scope, but its draw buffer is on, so its texture is cleared through a promoted clear. + /// Inside an open pass the clear stays vkCmdClearAttachments and is counted (returns + /// true with the scope open). With no pass open a full-mask clear is promoted into the + /// next scope attaching the image (returns false); a partial glColorMask clear opens + /// the scope and clears in it, as before. + /// + private bool ClearColorOnGraph(CommandBuffer commandBuffer, int attachment, float r, float g, float b, float a) + { + VulkanFramebuffer target = _bound!; + if (!InScope(target, attachment)) + { + // Left out by the declared pass while its draw buffer is on: GL clears the + // texture, so the clear is promoted and lands before the texture's next use. + if (((target.PassExclusion >> attachment) & 1) != 0) + { + VulkanTexture? excluded = _textures.Get(target.Color[attachment].TextureId); + if (excluded != null) + { + _graph.PromoteColorClear(excluded, target.Color[attachment].Layer, r, g, b, a); + } + } + return false; + } + + if (!_renderingActive || _needsRestart) + { + VulkanTexture? texture = _textures.Get(target.Color[attachment].TextureId); + if (texture == null) return false; + EndRendering(commandBuffer); + _graph.PromoteColorClear(texture, target.Color[attachment].Layer, r, g, b, a); + return false; + } + + _graph.NoteInPassClear(); + return true; + } + + public void ClearDepth(CommandBuffer commandBuffer, float depth) + { + if (_bound == null || _bound.DepthTextureId <= 0) return; + + if (_graph.Enabled) + { + VulkanTexture? texture = _textures.Get(_bound.DepthTextureId); + if (texture == null) return; + if (!_renderingActive || _needsRestart || DepthReadOnly) + { + // No pass open (or the scope is about to change): LOAD_OP_CLEAR on the next scope. + EndRendering(commandBuffer); + SetDepthReadOnly(false); + _graph.PromoteDepthClear(texture, depth); + return; + } + _graph.NoteInPassClear(); + } + + // A read-only depth attachment cannot be cleared; a clear is a write. + SetDepthReadOnly(false); + if (!_graph.Enabled) EnsureRendering(commandBuffer); + if (!_renderingActive) return; + + var clear = new ClearAttachment + { + AspectMask = ImageAspectFlags.DepthBit, + ClearValue = new ClearValue(depthStencil: new ClearDepthStencilValue(depth, 0)), + }; + var rect = new ClearRect + { + Rect = new Rect2D(new Offset2D(0, 0), new Extent2D(_bound.Width, _bound.Height)), + BaseArrayLayer = 0, + LayerCount = 1, + }; + _context.Api.CmdClearAttachments(commandBuffer, 1, &clear, 1, &rect); + } + + // ------------------------------------------------------------------ formats + + /// + /// The attachment formats of the bound target, for the pipeline key. They + /// follow the scope, not the draw buffers, so a mask toggle keeps the id: an + /// unbound or sample-excluded slot reports , + /// agreeing with the null attachment the scope was opened with. + /// + public int FormatsIdOf(VulkanFramebuffer framebuffer) + { + if (framebuffer.FormatsId >= 0) return framebuffer.FormatsId; + + int count = HighestScopeAttachment(framebuffer) + 1; + var colorFormats = new Format[Math.Max(count, 0)]; + + for (int i = 0; i < count; i++) + { + AttachmentSlot slot = framebuffer.Color[i]; + VulkanTexture? texture = InScope(framebuffer, i) ? _textures.Get(slot.TextureId) : null; + colorFormats[i] = texture?.Format ?? Format.Undefined; + } + + Format depthFormat = Format.Undefined; + if (framebuffer.DepthTextureId > 0) + { + depthFormat = _textures.Get(framebuffer.DepthTextureId)?.Format ?? Format.Undefined; + } + + framebuffer.FormatsId = _formats.Intern(new RenderTargetFormats(colorFormats, depthFormat)); + return framebuffer.FormatsId; + } + + /// The formats behind an id handed out. + public RenderTargetFormats FormatsOf(int formatsId) => _formats.Get(formatsId); + + /// + /// The attachment formats of the scope opens now, without + /// interning them: what a native draw checks its pipeline against. + /// + public RenderTargetFormats ScopeFormats(VulkanFramebuffer framebuffer) => + DeclaredFormats(framebuffer, ~framebuffer.PassExclusion); + + /// + /// The attachment formats of the scope a pass declared with opens + /// on : every bound slot among them, whatever pass the target + /// is in now (a native system builds its pipeline before its pass is declared). + /// + public RenderTargetFormats DeclaredFormats(VulkanFramebuffer framebuffer, uint colorSlots) + { + int count = 0; + for (int i = 0; i < RenderLimits.MaxColorAttachments; i++) + { + if (framebuffer.Color[i].IsBound && ((colorSlots >> i) & 1) != 0) count = i + 1; + } + + var colorFormats = new Format[count]; + for (int i = 0; i < count; i++) + { + bool inScope = framebuffer.Color[i].IsBound && ((colorSlots >> i) & 1) != 0; + VulkanTexture? texture = inScope ? _textures.Get(framebuffer.Color[i].TextureId) : null; + colorFormats[i] = texture?.Format ?? Format.Undefined; + } + + Format depthFormat = framebuffer.DepthTextureId > 0 + ? _textures.Get(framebuffer.DepthTextureId)?.Format ?? Format.Undefined + : Format.Undefined; + return new RenderTargetFormats(colorFormats, depthFormat); + } + + /// Colour attachments of the scope the framebuffer opens (highest participating slot + 1). + public int EnabledAttachmentCount(VulkanFramebuffer framebuffer) => + HighestScopeAttachment(framebuffer) + 1; + + private static int HighestScopeAttachment(VulkanFramebuffer framebuffer) + { + int highest = -1; + for (int i = 0; i < RenderLimits.MaxColorAttachments; i++) + { + if (InScope(framebuffer, i)) highest = i; + } + return highest; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + _framebuffers.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Core/RenderTrace.cs b/Optimum.Render.Vulkan/Core/RenderTrace.cs new file mode 100644 index 00000000..aed8d51a --- /dev/null +++ b/Optimum.Render.Vulkan/Core/RenderTrace.cs @@ -0,0 +1,127 @@ +using System; +using System.Globalization; +using System.IO; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// A trace of what the device was actually asked to draw, for the cases where +/// the frame is legal - the validation layer says nothing - but wrong. +/// +/// Off unless OPTIMUM_RENDER_TRACE names a file, so it costs one static bool +/// check in a release build and never appears in a normal session. It is a +/// debugging aid rather than diagnostics the game consumes: the client's own +/// error channel carries validation messages already. +/// +internal static class RenderTrace +{ + private static readonly object Gate = new(); + private static readonly string? Path = Environment.GetEnvironmentVariable("OPTIMUM_RENDER_TRACE"); + + public static bool Enabled => Path != null; + + public static void Write(string line) + { + if (Path == null) return; + lock (Gate) + { + File.AppendAllText(Path, line + "\n"); + } + } + + /// + /// Records a texture upload along with a checksum of its first rows, which + /// is what distinguishes "the image never got the pixels" from "the image is + /// correct but never sampled". + /// + public static unsafe void TextureCreated( + int id, int width, int height, Format format, IntPtr pixels, int bytesPerPixel) + { + if (Path == null) return; + + long sum = 0; + int nonZero = 0; + if (pixels != IntPtr.Zero && bytesPerPixel > 0) + { + int sampled = Math.Min(width * height * bytesPerPixel, 64 * 1024); + byte* bytes = (byte*)pixels; + for (int i = 0; i < sampled; i++) + { + sum += bytes[i]; + if (bytes[i] != 0) nonZero++; + } + } + + Write(string.Format(CultureInfo.InvariantCulture, + "tex create id={0} {1}x{2} format={3} bpp={4} bytesum={5} nonzero={6}", + id, width, height, format, bytesPerPixel, sum, nonZero)); + } + + /// + /// Dumps one named uniform out of a program's shadow buffer, which is what + /// separates "the CPU wrote nonsense" from "the CPU was right and the GPU + /// read it from the wrong place". + /// + public static void Uniforms(Shaders.ProgramInterfaceLayout layout, byte[] shadow, string name) + { + if (Path == null) return; + if (!layout.MembersByName.TryGetValue(name, out Shaders.UniformMember? member)) return; + + int floats = Math.Min(member.Size / sizeof(float), 16); + var text = new System.Text.StringBuilder(); + text.Append(" ").Append(name).Append(" @").Append(member.Offset).Append(" ="); + for (int i = 0; i < floats; i++) + { + text.Append(' ').Append( + BitConverter.ToSingle(shadow, member.Offset + i * sizeof(float)) + .ToString("0.###", CultureInfo.InvariantCulture)); + } + Write(text.ToString()); + } + + public static void UniformInt(Shaders.ProgramInterfaceLayout layout, byte[] shadow, string name) + { + if (Path == null) return; + if (!layout.MembersByName.TryGetValue(name, out Shaders.UniformMember? member)) return; + + Write(" " + name + " @" + member.Offset + " = " + BitConverter.ToInt32(shadow, member.Offset)); + } + + /// + /// Writes each stage's rewritten GLSL beside the trace file, so the source + /// the driver actually compiled can be read rather than reconstructed. + /// + public static void DumpProgramSources(string passName, Shaders.TranslatedProgram translated) + { + if (Path == null) return; + + string directory = System.IO.Path.GetDirectoryName(Path) ?? "."; + // The hardcoded minimal-GUI program has no pass name at all. + string safeName = string.IsNullOrWhiteSpace(passName) + ? "unnamed" + : string.Join("_", passName.Split(System.IO.Path.GetInvalidFileNameChars())); + + foreach (var stage in translated.RewrittenSource) + { + string file = System.IO.Path.Combine(directory, "shader-" + safeName + "-" + stage.Key + ".glsl"); + try + { + File.WriteAllText(file, stage.Value); + } + catch (IOException) + { + // Losing a debug dump must not disturb the run. + } + } + } + + public static void Draw(int meshId, int programId, int indexCount, bool depthTest, bool blend, float depthRangeHint) + { + if (Path == null) return; + + Write(string.Format(CultureInfo.InvariantCulture, + "draw mesh={0} program={1} indices={2} depthTest={3} blend={4} z={5}", + meshId, programId, indexCount, depthTest, blend, depthRangeHint)); + } +} diff --git a/Optimum.Render.Vulkan/Core/ShaderProgramResources.cs b/Optimum.Render.Vulkan/Core/ShaderProgramResources.cs new file mode 100644 index 00000000..24c25498 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/ShaderProgramResources.cs @@ -0,0 +1,314 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Core.Native; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Everything the GPU needs for one linked shader program: the modules and the +/// CPU-side shadow of its program record. Every program's pipelines are built +/// against the one shared pipeline layout (plan decision 9, ), +/// so a program owns no set layouts and no pipeline layout of its own. +/// +/// The shadow buffer is what makes GL's uniform protocol work. The game sets +/// uniforms one at a time by name, at any point before a draw, and expects the +/// values to persist for the life of the program. So writes land in this buffer, +/// and a draw copies it into the frame's uniform ring only when something +/// changed since the snapshot it last took. +/// +/// Values every program shares - fog, light, shadow cascades, warp, sky - are not +/// in this buffer at all: they live in the device's frame block (set 0, +/// ), and their locations point there. +/// +internal sealed unsafe class ShaderProgramResources : IDisposable +{ + private readonly VulkanContext _context; + private bool _disposed; + + public int ProgramId { get; } + public ProgramInterfaceLayout Interface { get; } + + public Dictionary Modules { get; } = new(); + + /// The shared pipeline layout this program's pipelines are created against. Not owned. + public PipelineLayout PipelineLayout { get; } + + /// + /// A shared layout of the program's own, for a program built outside a device (tests): + /// the same shape as the device's, set 1 included. Null for a device's programs. + /// + public SharedPipelineLayout? StandaloneLayout { get; } + + /// CPU mirror of the program record. + public byte[] UniformShadow { get; } + + /// + /// CPU mirror of a native program's push block when it holds members besides sampler slots; null + /// otherwise. A draw copies it into the device's push shadow before the slots are resolved over it. + /// + public byte[]? PushShadow { get; } + + /// The specialization constants every pipeline of this program is created with; null for a rewritten program. + public NativeSpecialization? Specialization { get; } + + /// Whether the program was linked from the native manifest. + public bool IsNative { get; } + + /// Bumped by every write that changes the shadow. + public uint UniformVersion { get; private set; } = 1; + + /// Which frame's ring holds the last snapshot of the shadow, taken at which version, where. + public uint SnapshotFrame { get; private set; } + public uint SnapshotVersion { get; private set; } + public uint SnapshotOffset { get; private set; } + + /// + /// Which texture unit each sampler uniform points at. In GL this is just an + /// int uniform; here it is the link between a bound texture and a descriptor. + /// + private readonly Dictionary _samplerIndices = new(StringComparer.Ordinal); + + /// Mutable unit assignments in sampler declaration order. + internal int[] SamplerUnitsByIndex { get; } + + /// Declaration-order names, fixed for this linked program. + public string[] SamplerNames { get; } + + /// + /// The device's shared pipeline layout. Programs built outside a device - in + /// tests - pass none and get a of the same shape. + /// + public ShaderProgramResources( + VulkanContext context, int programId, TranslatedProgram translated, PipelineLayout sharedLayout = default) + { + _context = context; + ProgramId = programId; + Interface = translated.Layout; + UniformShadow = translated.Layout.CreateShadowBuffer(); + PushShadow = translated.Layout.CreatePushShadow(); + Specialization = translated.Specialization; + IsNative = translated.IsNative; + + foreach (KeyValuePair stage in translated.Spirv) + { + Modules[stage.Key] = CreateModule(stage.Value); + } + SourceHash = HashSpirv(translated.Spirv, translated.Specialization); + + // Sampler uniforms default to the unit matching their declaration order, + // which is the order the game's own texture-location bookkeeping assigns. + SamplerNames = new string[Interface.Samplers.Count]; + SamplerUnitsByIndex = new int[Interface.Samplers.Count]; + for (int i = 0; i < Interface.Samplers.Count; i++) + { + SamplerBinding sampler = Interface.Samplers[i]; + SamplerNames[i] = sampler.Name; + _samplerIndices[sampler.Name] = i; + SamplerUnitsByIndex[i] = sampler.Order; + } + + if (sharedLayout.Handle == 0) + { + StandaloneLayout = SharedPipelineLayout.CreateStandalone(context); + sharedLayout = StandaloneLayout.Layout; + } + PipelineLayout = sharedLayout; + } + + /// + /// A hash of every stage's SPIR-V, in stage order, and of a native program's specialization + /// data. The rewriter's SPIR-V has its defines resolved; a native module's settings are its + /// specialization constants, so they are part of the program's identity. Two programs with + /// this hash build the same pipelines for the same state - what the pipeline-key log matches + /// on across launches. + /// + public UInt128 SourceHash { get; } + + private static UInt128 HashSpirv(Dictionary spirv, NativeSpecialization? specialization) + { + using var hash = System.Security.Cryptography.IncrementalHash.CreateHash( + System.Security.Cryptography.HashAlgorithmName.SHA256); + var stages = new List(spirv.Keys); + stages.Sort(); + Span header = stackalloc byte[8]; + foreach (EnumShaderType stage in stages) + { + BitConverter.TryWriteBytes(header, (int)stage); + BitConverter.TryWriteBytes(header[4..], spirv[stage].Length); + hash.AppendData(header); + hash.AppendData(spirv[stage]); + } + if (specialization != null) + { + foreach (NativeSpecialization.Entry entry in specialization.Entries) + { + BitConverter.TryWriteBytes(header, entry.Id); + BitConverter.TryWriteBytes(header[4..], entry.Offset); + hash.AppendData(header); + } + hash.AppendData(specialization.Data); + } + Span digest = stackalloc byte[32]; + hash.GetHashAndReset(digest); + return new UInt128(BitConverter.ToUInt64(digest[..8]), BitConverter.ToUInt64(digest.Slice(8, 8))); + } + + private ShaderModule CreateModule(byte[] spirv) + { + fixed (byte* code = spirv) + { + var createInfo = new ShaderModuleCreateInfo + { + SType = StructureType.ShaderModuleCreateInfo, + CodeSize = (nuint)spirv.Length, + PCode = (uint*)code, + }; + + if (_context.Api.CreateShaderModule(_context.Device, &createInfo, null, out ShaderModule module) + != Result.Success) + { + throw new InvalidOperationException("vkCreateShaderModule failed"); + } + return module; + } + } + + // ------------------------------------------------------------------ uniforms + + /// + /// The first sampler location. Sampler locations run downwards from here so + /// they can never collide with a uniform block offset, which is always zero + /// or positive, nor with GL's "not found" answer of -1. + /// + private const int FirstSamplerLocation = -2; + + /// + /// Where frame block locations start: the member's offset in the shared block + /// plus this. Far above any per-program block, so the two ranges never meet. + /// + public const int FrameLocationBase = 1 << 28; + + /// + /// Where a native program's push-member locations start: the member's offset in the push + /// block plus this. Above any record offset and below . + /// + public const int PushLocationBase = 1 << 27; + + /// Whether a location handed out by is a push-block member. + public static bool IsPushLocation(int location) => location >= PushLocationBase && location < FrameLocationBase; + + /// Whether a location handed out by names a sampler. + public static bool IsSamplerLocation(int location) => location <= FirstSamplerLocation; + + /// Whether a location handed out by is in the shared frame block. + public static bool IsFrameLocation(int location) => location >= FrameLocationBase; + + private static int SamplerIndexOf(int location) => FirstSamplerLocation - location; + + /// + /// Resolves a uniform name to an opaque location, the way glGetUniformLocation + /// does. + /// + /// Samplers are not members of the program record - they are push slots or + /// frame textures - but the client looks every declared uniform up by name and + /// treats a -1 as "the shader does not use this". Returning -1 for samplers + /// would tell it that every texture uniform in the game is unused, so they + /// get locations of their own from a disjoint range. Members of the shared + /// frame block get a third range. + /// + public int LocationOf(string name) + { + if (Interface.FrameMemberDeclaredLengths.ContainsKey(name) && + FrameGlobals.TryGetMember(name, out UniformMember frameMember)) + { + return FrameLocationBase + frameMember.Offset; + } + + if (Interface.PushMembersByName.TryGetValue(name, out UniformMember? pushMember)) + { + return PushLocationBase + pushMember.Offset; + } + + if (Interface.MembersByName.TryGetValue(name, out UniformMember? member)) + { + return member.Offset; + } + + for (int i = 0; i < Interface.Samplers.Count; i++) + { + if (string.Equals(Interface.Samplers[i].Name, name, StringComparison.Ordinal)) + { + return FirstSamplerLocation - i; + } + } + return -1; + } + + /// + /// Points the sampler at at a texture unit. + /// + /// GL assigns a sampler's unit by writing an int to its uniform location, so + /// a client that resolved a location and set it as an int lands here rather + /// than writing into the uniform block. + /// + public void SetSamplerUnitByLocation(int location, int unit) + { + int index = SamplerIndexOf(location); + if (index < 0 || index >= Interface.Samplers.Count) return; + + SamplerUnitsByIndex[index] = unit; + } + + public void SetSamplerUnitByName(string name, int unit) + { + if (_samplerIndices.TryGetValue(name, out int index)) + SamplerUnitsByIndex[index] = unit; + } + + /// Writes raw bytes at an offset previously handed out by . + public void SetUniform(int offset, ReadOnlySpan data) + { + if (offset < 0 || offset + data.Length > UniformShadow.Length) return; + + Span destination = UniformShadow.AsSpan(offset, data.Length); + if (data.SequenceEqual(destination)) return; + + data.CopyTo(destination); + UniformVersion++; + } + + /// Writes raw bytes at a push location previously handed out by . + public void SetPushUniform(int location, ReadOnlySpan data) + { + int offset = location - PushLocationBase; + if (PushShadow == null || offset < 0 || offset + data.Length > PushShadow.Length) return; + data.CopyTo(PushShadow.AsSpan(offset, data.Length)); + } + + /// Whether the shadow's current contents already sit in 's ring. + public bool HasSnapshotFor(uint frame) => SnapshotFrame == frame && SnapshotVersion == UniformVersion; + + public void NoteSnapshot(uint frame, uint offset) + { + SnapshotFrame = frame; + SnapshotVersion = UniformVersion; + SnapshotOffset = offset; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + Vk api = _context.Api; + // A device's shared layout belongs to the device. + StandaloneLayout?.Dispose(); + foreach (ShaderModule module in Modules.Values) + { + api.DestroyShaderModule(_context.Device, module, null); + } + } +} diff --git a/Optimum.Render.Vulkan/Core/SharedPipelineLayout.cs b/Optimum.Render.Vulkan/Core/SharedPipelineLayout.cs new file mode 100644 index 00000000..9bf4a4bf --- /dev/null +++ b/Optimum.Render.Vulkan/Core/SharedPipelineLayout.cs @@ -0,0 +1,205 @@ +using System; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Plan decision 9's one pipeline layout, shared by every program once shaders +/// target it (): +/// +/// | set 0 | FrameGlobals dynamic UBO and the fixed frame textures; a normal set (dynamic buffers cannot be update-after-bind) | +/// | set 1 | the bindless texture table's layout (), not owned here | +/// | set 2 | FaceData, the animation blocks, the program record (dynamic) and the named-block range, a normal set built per draw | +/// | push | for vertex and fragment | +/// +/// Created once at device bring-up; destroyed at teardown after the device-idle +/// wait, before the table's set layout it names. +/// +internal sealed unsafe class SharedPipelineLayout : IDisposable +{ + public const ShaderStageFlags Stages = ShaderStageFlags.VertexBit | ShaderStageFlags.FragmentBit; + + private readonly VulkanContext _context; + private readonly bool _ownsTextureSetLayout; + private bool _disposed; + + public DescriptorSetLayout FrameSetLayout { get; } + public DescriptorSetLayout TextureSetLayout { get; } + public DescriptorSetLayout StorageSetLayout { get; } + public PipelineLayout Layout { get; } + + /// + /// A layout of the same shape with a set 1 layout of its own, which it owns: for + /// programs built outside a device (tests). Pipelines built against it are + /// compatible with any set 1 layout built from the same capacities. + /// + public static SharedPipelineLayout CreateStandalone(VulkanContext context) + { + uint[] capacities = BindlessKinds.ClampCapacities(context.Capabilities.DescriptorIndexing, + DescriptorIndexingFloor.FrameTextures); + DescriptorSetLayout textures = BindlessTextureTable.CreateSetLayout(context, capacities); + try + { + return new SharedPipelineLayout(context, textures, ownsTextureSetLayout: true); + } + catch + { + context.Api.DestroyDescriptorSetLayout(context.Device, textures, null); + throw; + } + } + + public SharedPipelineLayout(VulkanContext context, DescriptorSetLayout textureSetLayout) + : this(context, textureSetLayout, ownsTextureSetLayout: false) + { + } + + private SharedPipelineLayout(VulkanContext context, DescriptorSetLayout textureSetLayout, bool ownsTextureSetLayout) + { + _context = context; + _ownsTextureSetLayout = ownsTextureSetLayout; + TextureSetLayout = textureSetLayout; + Vk api = context.Api; + + DescriptorSetLayoutBinding[] frameBindings = FrameBindings(); + DescriptorSetLayoutBinding[] storageBindings = StorageBindings(); + + FrameSetLayout = CreateSetLayout(frameBindings, "set 0 (frame)"); + try + { + StorageSetLayout = CreateSetLayout(storageBindings, "set 2 (storage)"); + } + catch + { + api.DestroyDescriptorSetLayout(context.Device, FrameSetLayout, null); + throw; + } + + DescriptorSetLayout* setLayouts = stackalloc DescriptorSetLayout[SetConvention.SetCount]; + setLayouts[SetConvention.FrameSet] = FrameSetLayout; + setLayouts[SetConvention.TextureSet] = TextureSetLayout; + setLayouts[SetConvention.StorageSet] = StorageSetLayout; + var pushConstants = new PushConstantRange + { + StageFlags = Stages, + Offset = 0, + Size = SetConvention.PushConstantBytes, + }; + var layoutInfo = new PipelineLayoutCreateInfo + { + SType = StructureType.PipelineLayoutCreateInfo, + SetLayoutCount = SetConvention.SetCount, + PSetLayouts = setLayouts, + PushConstantRangeCount = 1, + PPushConstantRanges = &pushConstants, + }; + PipelineLayout layout; + Result result = api.CreatePipelineLayout(context.Device, &layoutInfo, null, &layout); + if (result != Result.Success) + { + api.DestroyDescriptorSetLayout(context.Device, StorageSetLayout, null); + api.DestroyDescriptorSetLayout(context.Device, FrameSetLayout, null); + VulkanResult.Check(result, "vkCreatePipelineLayout for the shared layout"); + } + Layout = layout; + } + + /// Set 0: the FrameGlobals dynamic UBO and the fixed frame textures. + internal static DescriptorSetLayoutBinding[] FrameBindings() + { + var bindings = new DescriptorSetLayoutBinding[1 + SetConvention.FrameTextures.Length]; + bindings[0] = new DescriptorSetLayoutBinding + { + Binding = (uint)SetConvention.FrameGlobalsBinding, + DescriptorType = DescriptorType.UniformBufferDynamic, + DescriptorCount = 1, + StageFlags = Stages, + }; + for (int i = 0; i < SetConvention.FrameTextures.Length; i++) + { + bindings[1 + i] = new DescriptorSetLayoutBinding + { + Binding = (uint)SetConvention.FrameTextures[i].Value, + DescriptorType = DescriptorType.CombinedImageSampler, + DescriptorCount = SetConvention.FrameTextures[i].Capacity, + StageFlags = Stages, + }; + } + return bindings; + } + + /// + /// Set 2: the storage buffers, the program record (a dynamic uniform buffer, + /// docs/vulkan.md) and the named-block range, where a + /// rewritten program's other named blocks sit as std140 storage buffers. Set 2 is + /// a normal set, so the dynamic buffer is legal beside the update-after-bind set 1; + /// the device floor counts it (). + /// + internal static DescriptorSetLayoutBinding[] StorageBindings() + { + int named = SetConvention.NamedBlockLastBinding - SetConvention.NamedBlockFirstBinding + 1; + var bindings = new DescriptorSetLayoutBinding[SetConvention.StorageBuffers.Length + 1 + named]; + int index = 0; + foreach (SetConvention.Binding buffer in SetConvention.StorageBuffers) + { + bindings[index++] = new DescriptorSetLayoutBinding + { + Binding = (uint)buffer.Value, + DescriptorType = DescriptorType.StorageBuffer, + DescriptorCount = buffer.Capacity, + StageFlags = Stages, + }; + } + bindings[index++] = new DescriptorSetLayoutBinding + { + Binding = (uint)SetConvention.ProgramRecordBinding, + DescriptorType = DescriptorType.UniformBufferDynamic, + DescriptorCount = 1, + StageFlags = Stages, + }; + for (int binding = SetConvention.NamedBlockFirstBinding; binding <= SetConvention.NamedBlockLastBinding; binding++) + { + bindings[index++] = new DescriptorSetLayoutBinding + { + Binding = (uint)binding, + DescriptorType = DescriptorType.StorageBuffer, + DescriptorCount = 1, + StageFlags = Stages, + }; + } + return bindings; + } + + /// The descriptor type set 2 declares at . + public static DescriptorType StorageSetDescriptorType(uint binding) => + binding == SetConvention.ProgramRecordBinding ? DescriptorType.UniformBufferDynamic : DescriptorType.StorageBuffer; + + private DescriptorSetLayout CreateSetLayout(DescriptorSetLayoutBinding[] bindings, string what) + { + fixed (DescriptorSetLayoutBinding* bindingsPtr = bindings) + { + var info = new DescriptorSetLayoutCreateInfo + { + SType = StructureType.DescriptorSetLayoutCreateInfo, + BindingCount = (uint)bindings.Length, + PBindings = bindingsPtr, + }; + DescriptorSetLayout layout; + VulkanResult.Check(_context.Api.CreateDescriptorSetLayout(_context.Device, &info, null, &layout), + "vkCreateDescriptorSetLayout for the shared layout's " + what); + return layout; + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + Vk api = _context.Api; + api.DestroyPipelineLayout(_context.Device, Layout, null); + api.DestroyDescriptorSetLayout(_context.Device, StorageSetLayout, null); + api.DestroyDescriptorSetLayout(_context.Device, FrameSetLayout, null); + if (_ownsTextureSetLayout) api.DestroyDescriptorSetLayout(_context.Device, TextureSetLayout, null); + } +} diff --git a/Optimum.Render.Vulkan/Core/TextureDump.cs b/Optimum.Render.Vulkan/Core/TextureDump.cs new file mode 100644 index 00000000..5ffecd50 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/TextureDump.cs @@ -0,0 +1,448 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.IO; +using System.Runtime.InteropServices; +using Silk.NET.Vulkan; +using Vintagestory.API.Config; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Writes a texture's contents, as they actually sit on the GPU, to a file. +/// +/// A wrong image and wrong texture coordinates look identical from the far end +/// of the pipeline, and the two can be argued about indefinitely. Reading the +/// image back settles it: whatever comes out is what every sample of that +/// texture saw, with no inference in between. +/// +/// Off unless OPTIMUM_DUMP_TEXTURES lists texture ids, comma separated. The +/// files land in OPTIMUM_DUMP_DIR when that names an absolute path, else beside +/// the render trace, else under the temp directory - never the working +/// directory - as binary PPM - a five-line header and raw RGB, which needs no encoder here +/// and which every image tool reads. +/// +internal static class TextureDump +{ + private static readonly string? Requested = + Environment.GetEnvironmentVariable("OPTIMUM_DUMP_TEXTURES"); + + /// Texture ids still waiting to be written. + private static readonly HashSet Pending = Parse(Requested); + + /// + /// Set by OPTIMUM_DUMP_TEXTURES=terrain, which asks for whatever the chunk + /// pass binds rather than for an id. + /// + /// Atlas ids are only handed out once a world loads, and are not stable + /// between runs, so naming one up front means guessing. Latching onto the + /// first storage-buffer multi-draw instead catches the block atlas at the + /// one moment it is certainly the texture the terrain is being drawn with. + /// + private static bool _wantsTerrain = + string.Equals(Requested?.Trim(), "terrain", StringComparison.OrdinalIgnoreCase); + + public static bool WantsTerrain => _wantsTerrain; + + /// Records the textures a chunk draw is using, and stops asking. + public static void RequestTerrain(int baseTexture, int linearTexture) + { + _wantsTerrain = false; + if (baseTexture > 0) Pending.Add(baseTexture); + if (linearTexture > 0 && linearTexture != baseTexture) Pending.Add(linearTexture); + } + + /// True while any requested texture has not been written yet. + /// + /// Frames to let pass before writing anything. OPTIMUM_DUMP_AFTER_FRAMES + /// (default 0) lets a dump of a frame target wait until a world is on + /// screen instead of capturing the menu's black first frame. + /// + private static readonly long StartAfterFrames = + long.TryParse(Environment.GetEnvironmentVariable("OPTIMUM_DUMP_AFTER_FRAMES"), NumberStyles.Integer, + CultureInfo.InvariantCulture, out long frames) ? frames : 0; + + /// + /// Seconds to wait before writing, OPTIMUM_DUMP_AFTER_SECONDS (default 0). + /// The menu runs uncapped, so a frame count alone can expire before a world + /// is on screen; wall time is what a person setting this reasons in. + /// + private static readonly double StartAfterSeconds = + double.TryParse(Environment.GetEnvironmentVariable("OPTIMUM_DUMP_AFTER_SECONDS"), NumberStyles.Float, + CultureInfo.InvariantCulture, out double seconds) ? seconds : 0; + + private static long _framesSeen; + private static readonly long StartedAt = System.Diagnostics.Stopwatch.GetTimestamp(); + + /// Counts a presented frame; the dump waits out the configured delays. + public static void NoteFrame() => _framesSeen++; + + private static double SecondsSinceStart => + (System.Diagnostics.Stopwatch.GetTimestamp() - StartedAt) / (double)System.Diagnostics.Stopwatch.Frequency; + + public static bool Wanted => + Pending.Count > 0 && _framesSeen >= StartAfterFrames && SecondsSinceStart >= StartAfterSeconds; + + private static HashSet Parse(string? value) + { + var ids = new HashSet(); + if (string.IsNullOrWhiteSpace(value)) return ids; + + foreach (string part in value.Split(',', StringSplitOptions.RemoveEmptyEntries)) + { + if (int.TryParse(part.Trim(), NumberStyles.Integer, CultureInfo.InvariantCulture, out int id)) + { + ids.Add(id); + } + } + return ids; + } + + /// + /// The ids still to write, as a snapshot safe to iterate while removing. + /// Ids stay pending until reports a successful write, + /// so a texture that does not exist yet or whose write fails is retried on a + /// later frame instead of being silently dropped. + /// + public static int[] Take() + { + var ids = new int[Pending.Count]; + Pending.CopyTo(ids); + return ids; + } + + /// Removes an id from the pending set once it has been written successfully. + public static void Complete(int textureId) => Pending.Remove(textureId); + + /// + /// Where the files go. An explicit OPTIMUM_DUMP_DIR must be absolute so + /// the launching environment names the location outright rather than + /// relative to whatever the working directory happens to be; otherwise + /// the files sit beside the render trace, and failing both, under a + /// dedicated folder in the temp directory. Never the working directory. + /// + private static string? Directory() + { + string? explicitDir = Environment.GetEnvironmentVariable("OPTIMUM_DUMP_DIR"); + if (!string.IsNullOrWhiteSpace(explicitDir)) + { + return Path.IsPathRooted(explicitDir) ? Path.GetFullPath(explicitDir) : null; + } + + string? tracePath = Environment.GetEnvironmentVariable("OPTIMUM_RENDER_TRACE"); + if (!string.IsNullOrWhiteSpace(tracePath) && Path.IsPathRooted(tracePath)) + { + string? beside = Path.GetDirectoryName(Path.GetFullPath(tracePath)); + if (!string.IsNullOrWhiteSpace(beside)) return beside; + } + + return Path.Combine(Path.GetTempPath(), "optimum-texture-dumps"); + } + + /// + /// One prefix per process, so two runs into the same directory never + /// overwrite each other's files and a run never overwrites its own. + /// + private static readonly string RunPrefix = + DateTime.UtcNow.ToString("yyyyMMdd-HHmmss", CultureInfo.InvariantCulture) + "-" + Environment.ProcessId; + + /// + /// Writes a texture's raw GPU bytes as a binary PPM, converting whatever + /// format the texture actually carries into 8-bit RGB. + /// + /// R16G16B16A16Sfloat and R32Sfloat are readback formats an attachment can + /// legitimately be dumped in (TAA motion, a depth-like target) rather than + /// the 8-bit RGBA/BGRA every other texture uses, so each gets its own + /// normalisation: + /// - Colour-shaped float data (R16G16B16A16Sfloat) is clamped to [0,1] and + /// scaled to a byte, same as any other colour channel. + /// - Single-channel float data (R32Sfloat and R16Sfloat, the latter read as + /// System.Half) is treated as motion-like and + /// mapped from [-64,64] pixels to [0,255], with 128 standing for zero + /// displacement - there is no separate "depth" convention to distinguish + /// it from motion at this format, so callers dumping true depth should + /// expect the same [-64,64]-centred-at-128 mapping. + /// + /// True if the file was written. + public static bool Write(int textureId, int width, int height, bool bgra, Format format, + ReadOnlySpan data) + { + if (width <= 0 || height <= 0) return false; + + int bytesPerPixel = BytesPerTexel(format); + if (data.Length < width * height * bytesPerPixel) return false; + + try + { + string? directory = Directory(); + if (directory == null) return false; + System.IO.Directory.CreateDirectory(directory); + string path = Path.Combine(directory, $"{RunPrefix}-texture-{textureId}-{width}x{height}.ppm"); + + // CreateNew: an existing file is never truncated, whatever named it. + using var file = new FileStream(path, FileMode.CreateNew, FileAccess.Write); + using var writer = new BinaryWriter(file); + + foreach (char c in $"P6\n{width} {height}\n255\n") writer.Write((byte)c); + + var row = new byte[width * 3]; + int stride = width * bytesPerPixel; + + switch (format) + { + case Format.R16G16B16A16Sfloat: + { + var floats = MemoryMarshal.Cast(data); + int floatsPerRow = width * 4; + for (int y = 0; y < height; y++) + { + var source = floats.Slice(y * floatsPerRow, floatsPerRow); + for (int x = 0; x < width; x++) + { + row[x * 3] = ColorByte((float)source[x * 4]); + row[x * 3 + 1] = ColorByte((float)source[x * 4 + 1]); + row[x * 3 + 2] = ColorByte((float)source[x * 4 + 2]); + } + writer.Write(row); + } + break; + } + case Format.R16Sfloat: + { + var halves = MemoryMarshal.Cast(data); + for (int y = 0; y < height; y++) + { + var source = halves.Slice(y * width, width); + for (int x = 0; x < width; x++) + { + byte value = MotionByte((float)source[x]); + row[x * 3] = value; + row[x * 3 + 1] = value; + row[x * 3 + 2] = value; + } + writer.Write(row); + } + break; + } + case Format.R32Sfloat: + { + var floats = MemoryMarshal.Cast(data); + for (int y = 0; y < height; y++) + { + var source = floats.Slice(y * width, width); + for (int x = 0; x < width; x++) + { + byte value = MotionByte(source[x]); + row[x * 3] = value; + row[x * 3 + 1] = value; + row[x * 3 + 2] = value; + } + writer.Write(row); + } + break; + } + case Format.R8Unorm or Format.R8Uint or Format.R8Srgb: + { + for (int y = 0; y < height; y++) + { + var source = data.Slice(y * stride, width); + for (int x = 0; x < width; x++) + { + byte value = source[x]; + row[x * 3] = value; + row[x * 3 + 1] = value; + row[x * 3 + 2] = value; + } + writer.Write(row); + } + break; + } + default: + { + int red = bgra ? 2 : 0; + int blue = bgra ? 0 : 2; + for (int y = 0; y < height; y++) + { + int source = y * stride; + for (int x = 0; x < width; x++) + { + row[x * 3] = data[source + x * 4 + red]; + row[x * 3 + 1] = data[source + x * 4 + 1]; + row[x * 3 + 2] = data[source + x * 4 + blue]; + } + writer.Write(row); + } + break; + } + } + + return true; + } + catch (IOException) + { + return false; + } + catch (UnauthorizedAccessException) + { + return false; + } + } + + /// + /// Bytes per texel for the formats the dump path is expected to see. One + /// table serves both the size check and the decode switch, so a format can + /// never be sized one way and read another; R16Sfloat sized as 4 bytes made + /// every row of an R16f readback start on the wrong texel. Anything + /// unrecognised falls back to 4 (8-bit RGBA), the blanket assumption the + /// default decode branch makes. + /// + public static int BytesPerTexel(Format format) => format switch + { + Format.R16G16B16A16Sfloat => 8, + Format.R16G16B16A16Unorm => 8, + Format.R32G32B32A32Sfloat => 16, + Format.R32Sfloat => 4, + Format.R16Sfloat => 2, + Format.D32Sfloat => 4, + Format.D16Unorm => 2, + Format.R8Unorm or Format.R8Uint or Format.R8Srgb => 1, + _ => 4, + }; + + /// + /// The GL token for a Vulkan format, for textures created without one + /// ( is 0). The inverse of + /// where that mapping is one to one. + /// + public static int GlInternalFormatOf(Format format) => format switch + { + Format.R8G8B8A8Unorm or Format.R8G8B8A8Srgb or Format.B8G8R8A8Unorm or Format.B8G8R8A8Srgb => 0x8058, + Format.R8Unorm => 0x8229, + Format.R16G16B16A16Sfloat => 0x881A, + Format.R16G16B16A16Unorm => 0x805B, + Format.R32G32B32A32Sfloat => 0x8814, + Format.R16Sfloat => 0x822D, + Format.R32Sfloat => 0x822E, + Format.B10G11R11UfloatPack32 => 0x8C3A, + Format.D32Sfloat => 0x8CAC, + Format.D16Unorm => 0x81A5, + _ => 0, + }; + + /// + /// Decodes a raw level-0 readback (, rows in memory + /// order, which is GL order) into the parity dump's shared representation - + /// what glGetTexImage returns on the OpenGL path: RGBA8 bytes for 8-bit + /// unsigned-normalised formats, RGBA float32 for other colour formats (missing + /// channels 0, alpha 1, as GL fills them), one float32 per texel for depth. + /// Returns null for a format the dump does not decode. + /// + public static OptimumTextureReadback? ToParityReadback(Format format, int glInternalFormat, + int width, int height, ReadOnlySpan data) + { + if (width <= 0 || height <= 0) return null; + int texels = width * height; + if (data.Length < texels * BytesPerTexel(format)) return null; + + var readback = new OptimumTextureReadback + { + GlInternalFormat = glInternalFormat, + Width = width, + Height = height, + }; + + switch (format) + { + case Format.R8G8B8A8Unorm or Format.R8G8B8A8Srgb: + readback.Bytes = data.Slice(0, texels * 4).ToArray(); + return readback; + case Format.B8G8R8A8Unorm or Format.B8G8R8A8Srgb: + { + var bytes = new byte[texels * 4]; + for (int i = 0; i < texels; i++) + { + bytes[i * 4] = data[i * 4 + 2]; + bytes[i * 4 + 1] = data[i * 4 + 1]; + bytes[i * 4 + 2] = data[i * 4]; + bytes[i * 4 + 3] = data[i * 4 + 3]; + } + readback.Bytes = bytes; + return readback; + } + case Format.R8Unorm: + { + var bytes = new byte[texels * 4]; + for (int i = 0; i < texels; i++) + { + bytes[i * 4] = data[i]; + bytes[i * 4 + 3] = 255; + } + readback.Bytes = bytes; + return readback; + } + case Format.R16G16B16A16Sfloat: + { + var source = MemoryMarshal.Cast(data); + var floats = new float[texels * 4]; + for (int i = 0; i < floats.Length; i++) floats[i] = (float)source[i]; + readback.Floats = floats; + return readback; + } + case Format.R16G16B16A16Unorm: + { + var source = MemoryMarshal.Cast(data); + var floats = new float[texels * 4]; + for (int i = 0; i < floats.Length; i++) floats[i] = source[i] / 65535f; + readback.Floats = floats; + return readback; + } + case Format.R32G32B32A32Sfloat: + readback.Floats = MemoryMarshal.Cast(data).Slice(0, texels * 4).ToArray(); + return readback; + case Format.R16Sfloat: + { + var source = MemoryMarshal.Cast(data); + var floats = new float[texels * 4]; + for (int i = 0; i < texels; i++) + { + floats[i * 4] = (float)source[i]; + floats[i * 4 + 3] = 1f; + } + readback.Floats = floats; + return readback; + } + case Format.R32Sfloat: + { + var source = MemoryMarshal.Cast(data); + var floats = new float[texels * 4]; + for (int i = 0; i < texels; i++) + { + floats[i * 4] = source[i]; + floats[i * 4 + 3] = 1f; + } + readback.Floats = floats; + return readback; + } + case Format.D32Sfloat: + readback.Floats = MemoryMarshal.Cast(data).Slice(0, texels).ToArray(); + return readback; + case Format.D16Unorm: + { + var source = MemoryMarshal.Cast(data); + var floats = new float[texels]; + for (int i = 0; i < texels; i++) floats[i] = source[i] / 65535f; + readback.Floats = floats; + return readback; + } + default: + return null; + } + } + + /// Clamps [0,1] colour data to a byte. + private static byte ColorByte(float value) => (byte)(Math.Clamp(value, 0f, 1f) * 255f); + + /// Maps [-64,64] px of motion-like data to [0,255], 128 = zero. + private static byte MotionByte(float value) => + (byte)Math.Clamp((value / 64f) * 127f + 128f, 0f, 255f); +} diff --git a/Optimum.Render.Vulkan/Core/TextureManager.cs b/Optimum.Render.Vulkan/Core/TextureManager.cs new file mode 100644 index 00000000..7942b399 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/TextureManager.cs @@ -0,0 +1,1026 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// The sampler state GL keeps on the texture object. +/// +/// In GL these live on the texture and are changed with glTexParameter; in Vulkan +/// they belong to a separate immutable sampler object. Keeping them here as a +/// value and resolving to a cached sampler at bind time reproduces the GL +/// behaviour without creating an object per texture. +/// +/// +/// Whether the GL min filter is one of the four MIPMAP forms. GL treats +/// GL_NEAREST and GL_LINEAR as "level 0 only" however many levels the texture +/// has, and Vulkan has no such filter - it always picks a level from the range +/// the sampler allows. So this decides the sampler's LOD clamp, and without it +/// a texture that merely owns a mip chain gets minified through it on surfaces +/// GL would have sampled sharp. +/// +/// +/// GL_TEXTURE_MAX_LEVEL, the highest mip the texture is allowed to use, or a +/// negative value for no limit. The client clamps this to the mipmap quality +/// setting after building a chain. +/// +internal readonly record struct SamplerState( + Filter MagFilter, + Filter MinFilter, + SamplerMipmapMode MipmapMode, + SamplerAddressMode AddressU, + SamplerAddressMode AddressV, + float LodBias, + bool CompareEnable, + float MaxAnisotropy, + BorderColor BorderColor, + bool Mipmapped = false, + int MaxLevel = -1) +{ + public static SamplerState Default => new( + Filter.Nearest, Filter.Nearest, SamplerMipmapMode.Nearest, + SamplerAddressMode.Repeat, SamplerAddressMode.Repeat, + 0f, false, 1f, BorderColor.FloatOpaqueBlack); + + /// + /// The sampler's LOD ceiling. Anything under 1 confines sampling to level 0, + /// which is what a non-mipmapping GL filter means. + /// + public float LodCeiling => !Mipmapped ? 0.25f + : MaxLevel >= 0 ? MaxLevel + : Vk.LodClampNone; +} + +/// A texture, its memory, its view, and the GL state attached to it. +internal sealed unsafe class VulkanTexture : IDisposable +{ + private readonly VulkanContext _context; + private bool _disposed; + + public Image Image { get; init; } + public MemoryAllocation Allocation { get; init; } + public ImageView View { get; init; } + + /// Never reused, unlike ; see . + public ulong Id { get; } = ResourceIds.Next(); + + private volatile bool _released; + + /// + /// Set by under the upload lock, before the + /// texture's bindless slots are released; a released texture gets no new slot. + /// + public bool Released + { + get => _released; + internal set => _released = value; + } + + public Format Format { get; init; } + + /// + /// The GL internal format token the client asked for, or 0 when the texture + /// was not created through a GL-token entry point. Kept because the Vulkan + /// format can be a promotion (GL_RGB lands in RGBA8 storage), and the parity + /// dump names files by what was requested so both backends pair. + /// + public int GlInternalFormat { get; set; } + + public uint Width { get; init; } + public uint Height { get; init; } + public uint MipLevels { get; init; } + public uint Layers { get; init; } + + /// The image usage it was created with; 0 for images created outside 's paths. + public ImageUsageFlags Usage { get; init; } + + /// Whether the view is a cube rather than a six-layer array. + public bool Cube { get; init; } + + /// Whether the image is 3D (); GL-created textures never are. + public bool Volume { get; init; } + public ImageAspectFlags Aspect { get; init; } + + /// Mutable, as glTexParameter is. + public SamplerState State { get; set; } = SamplerState.Default; + + private ResourceStateTracker? _sync; + + /// + /// Per-subresource layout, write and read stages, which every barrier on this + /// texture derives from (). Created on first use, + /// once the image's dimensions are set. + /// + internal ResourceStateTracker Sync + { + get + { + ResourceStateTracker? sync = _sync; + if (sync != null) return sync; + System.Threading.Interlocked.CompareExchange(ref _sync, + new ResourceStateTracker(MipLevels, Layers, Aspect != ImageAspectFlags.ColorBit), null); + return _sync!; + } + } + + /// + /// The layout of the whole image, tracked because Vulkan offers no way to + /// query it; UNDEFINED while its subresources are in different layouts. + /// + public ImageLayout Layout => Sync.Layout; + + /// + /// The frame command buffer generation that last used this texture; an + /// upload to a texture the frame being recorded already used goes inline. + /// See . + /// + internal long FrameUse; + + /// + /// Single-layer views, created on demand and keyed by layer. + /// + /// covers the whole image, which is what a sampler wants. + /// A colour attachment pointed at one layer of an array needs a view of that + /// layer alone - the OIT accumulation target is one array attached three + /// times, once per layer, and a whole-image view there sends all three + /// attachments to the same layer. + /// + private readonly Dictionary _layerViews = new(); + + public VulkanTexture(VulkanContext context) => _context = context; + + public ImageView ViewOfLayer(uint layer) + { + if (layer == 0 && Layers <= 1) return View; + if (_layerViews.TryGetValue(layer, out ImageView existing)) return existing; + + var createInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = Image, + ViewType = ImageViewType.Type2D, + Format = Format, + SubresourceRange = new ImageSubresourceRange(Aspect, 0, MipLevels, layer, 1), + }; + + if (_context.Api.CreateImageView(_context.Device, &createInfo, null, out ImageView view) != Result.Success) + { + return View; + } + _layerViews[layer] = view; + return view; + } + + /// + /// 2D views of a mip range, created on demand. A storage image descriptor names + /// exactly one level, and a sampled read of level n must not name the levels the + /// same pass writes: validation checks the layout of every level a view covers. + /// + private readonly Dictionary<(uint BaseMip, uint MipCount), ImageView> _mipViews = new(); + + public ImageView ViewOfMips(uint baseMip, uint mipCount) + { + if (baseMip == 0 && mipCount >= MipLevels && Layers <= 1 && !Cube && !Volume) return View; + lock (_mipViews) + { + if (_mipViews.TryGetValue((baseMip, mipCount), out ImageView existing)) return existing; + + var createInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = Image, + ViewType = ImageViewType.Type2D, + Format = Format, + SubresourceRange = new ImageSubresourceRange(Aspect, baseMip, mipCount, 0, 1), + }; + VulkanResult.Check(_context.Api.CreateImageView(_context.Device, &createInfo, null, out ImageView view), + "vkCreateImageView for mips " + baseMip + "+" + mipCount); + _mipViews[(baseMip, mipCount)] = view; + return view; + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + Vk api = _context.Api; + foreach (ImageView layerView in _layerViews.Values) + { + if (layerView.Handle != 0) api.DestroyImageView(_context.Device, layerView, null); + } + _layerViews.Clear(); + foreach (ImageView mipView in _mipViews.Values) + { + if (mipView.Handle != 0) api.DestroyImageView(_context.Device, mipView, null); + } + _mipViews.Clear(); + if (View.Handle != 0) api.DestroyImageView(_context.Device, View, null); + if (Image.Handle != 0) api.DestroyImage(_context.Device, Image, null); + if (Allocation.IsValid) _context.Allocator.Free(Allocation); + } +} + +/// +/// Interns sampler objects by their state. +/// +/// The game has a handful of distinct sampler configurations - nearest and linear, +/// clamped and repeating, plus the shadow-comparison and mip-bias variants - but +/// sets them on hundreds of textures. One object per distinct state rather than +/// per texture keeps the count in single digits. +/// +internal sealed unsafe class SamplerCache : IDisposable +{ + private readonly VulkanContext _context; + private readonly Dictionary _samplers = new(); + private bool _disposed; + + public int Count => _samplers.Count; + + public SamplerCache(VulkanContext context) => _context = context; + + public Sampler Get(SamplerState state) + { + if (_samplers.TryGetValue(state, out Sampler existing)) return existing; + + float maxAnisotropy = _context.Capabilities.SamplerAnisotropy + ? Math.Max(1f, state.MaxAnisotropy) + : 1f; + + var createInfo = new SamplerCreateInfo + { + SType = StructureType.SamplerCreateInfo, + MagFilter = state.MagFilter, + MinFilter = state.MinFilter, + MipmapMode = state.MipmapMode, + AddressModeU = state.AddressU, + AddressModeV = state.AddressV, + AddressModeW = SamplerAddressMode.ClampToEdge, + MipLodBias = Math.Clamp(state.LodBias, -_context.Capabilities.MaxSamplerLodBias, + _context.Capabilities.MaxSamplerLodBias), + AnisotropyEnable = maxAnisotropy > 1f, + MaxAnisotropy = maxAnisotropy, + CompareEnable = state.CompareEnable, + // Shadow maps sample with a less-or-equal comparison, matching the + // GL_COMPARE_REF_TO_TEXTURE mode the shadow passes enable. + CompareOp = CompareOp.LessOrEqual, + MinLod = 0f, + MaxLod = state.LodCeiling, + BorderColor = state.BorderColor, + UnnormalizedCoordinates = false, + }; + + if (_context.Api.CreateSampler(_context.Device, &createInfo, null, out Sampler sampler) != Result.Success) + { + throw new InvalidOperationException("vkCreateSampler failed"); + } + + _samplers[state] = sampler; + return sampler; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + foreach (Sampler sampler in _samplers.Values) + { + _context.Api.DestroySampler(_context.Device, sampler, null); + } + _samplers.Clear(); + } +} + +/// +/// Owns every texture and hands out integer ids in place of GL names. +/// +/// The ids have to stay integers because the game's public API exposes them: +/// LoadedTexture.TextureId and FrameBufferRef.ColorTextureIds are +/// fields mods read and pass back. So this is a handle table, and 0 means "no +/// texture" exactly as it does in GL. +/// +internal sealed unsafe class TextureManager : IDisposable +{ + /// GL_SHORT source pixels converted to GL_RGBA16 storage. + internal static ushort ShortToUnorm16(short value) => + (ushort)((Math.Max(0, (int)value) * 65535L + 16383) / 32767); + + public void UploadNormalizedShorts(int id, int level, int x, int y, + int width, int height, ReadOnlySpan pixels) + { + int count = checked(width * height * 4); + var converted = new ushort[count]; + for (int i = 0; i < count; i++) converted[i] = ShortToUnorm16(pixels[i]); + + fixed (ushort* source = converted) + { + Upload(id, level, x, y, (uint)width, (uint)height, (IntPtr)source, 8); + } + } + + private readonly VulkanContext _context; + private readonly UploadManager _uploads; + private readonly List _textures = new(); + private readonly Stack _freeIds = new(); + private bool _disposed; + + public SamplerCache Samplers { get; } + + public TextureManager(VulkanContext context, UploadManager uploads) + { + _context = context; + _uploads = uploads; + Samplers = new SamplerCache(context); + _barriers = CreateBatcher(); + + // Index 0 is reserved so a zero id never names a real texture. + _textures.Add(null); + } + + public int Count + { + get + { + int live = 0; + foreach (VulkanTexture? texture in _textures) + { + if (texture != null) live++; + } + return live; + } + } + + public VulkanTexture? Get(int id) => + id > 0 && id < _textures.Count ? _textures[id] : null; + + private int Register(VulkanTexture texture) + { + // Under the upload lock, like Delete: an upload from another thread + // looks its texture up again under the same lock. + _uploads.EnterLock(); + try + { + if (_freeIds.Count > 0) + { + int reused = _freeIds.Pop(); + _textures[reused] = texture; + return reused; + } + + _textures.Add(texture); + return _textures.Count - 1; + } + finally + { + _uploads.ExitLock(); + } + } + + /// + /// Creates a texture. Usage always includes transfer source and destination + /// so uploads, readback and mipmap generation need no advance warning, which + /// is the GL model where any texture can be updated at any time. + /// + public int Create( + uint width, uint height, Format format, + uint layers = 1, bool cube = false, bool generateMipmaps = false, + ImageUsageFlags extraUsage = 0, MemoryPoolClass poolClass = MemoryPoolClass.DeviceImages) + { + // GL tolerates a zero-sized texture - it creates nothing and carries on - + // while Vulkan rejects the extent outright. The client asks for one when + // a render target is sized from a window dimension that is still zero, so + // this clamps rather than throwing, matching GL's forgiveness. + width = Math.Max(1, width); + height = Math.Max(1, height); + + uint mipLevels = generateMipmaps ? MipLevelsFor(width, height) : 1; + ImageAspectFlags aspect = IsDepthFormat(format) + ? ImageAspectFlags.DepthBit + : ImageAspectFlags.ColorBit; + + ImageUsageFlags usage = + ImageUsageFlags.SampledBit | ImageUsageFlags.TransferDstBit | ImageUsageFlags.TransferSrcBit + | extraUsage + | (IsDepthFormat(format) + ? ImageUsageFlags.DepthStencilAttachmentBit + : ImageUsageFlags.ColorAttachmentBit); + + return CreateImage(width, height, format, mipLevels, layers, cube, usage, aspect, poolClass); + } + + /// + /// Creates a texture a compute pass writes as a storage image (and later passes + /// sample): when the device can store to and sample + /// it, otherwise the first wider format of the same kind that it can + /// (, ending in RGBA8). Transfer usage is + /// always included, as for every texture; a colour attachment usage only where + /// the chosen format supports it. is explicit: a + /// prefiltered chain keeps as many levels as its consumer reads, not a full chain. + /// + public int CreateStorage(uint width, uint height, Format requested, uint mipLevels = 1, + MemoryPoolClass poolClass = MemoryPoolClass.DeviceImages) + { + width = Math.Max(1, width); + height = Math.Max(1, height); + mipLevels = Math.Clamp(mipLevels, 1, MipLevelsFor(width, height)); + + Format format = StorageFormats.Choose(requested, _context.OptimalFormatFeatures); + if (format != requested) + { + RenderTrace.Write("storage texture: " + requested + " is not storage-capable here; using " + format); + } + + ImageUsageFlags usage = ImageUsageFlags.StorageBit | ImageUsageFlags.SampledBit | + ImageUsageFlags.TransferDstBit | ImageUsageFlags.TransferSrcBit; + if (StorageFormats.SupportsColorAttachment(_context.OptimalFormatFeatures(format))) + { + usage |= ImageUsageFlags.ColorAttachmentBit; + } + + return CreateImage(width, height, format, mipLevels, 1, false, usage, ImageAspectFlags.ColorBit, poolClass); + } + + private int CreateImage(uint width, uint height, Format format, uint mipLevels, uint layers, bool cube, + ImageUsageFlags usage, ImageAspectFlags aspect, MemoryPoolClass poolClass) + { + var imageInfo = new ImageCreateInfo + { + SType = StructureType.ImageCreateInfo, + ImageType = ImageType.Type2D, + Format = format, + Extent = new Extent3D(width, height, 1), + MipLevels = mipLevels, + ArrayLayers = cube ? 6 : layers, + Samples = SampleCountFlags.Count1Bit, + Tiling = ImageTiling.Optimal, + Usage = usage, + SharingMode = SharingMode.Exclusive, + InitialLayout = ImageLayout.Undefined, + Flags = cube ? ImageCreateFlags.CreateCubeCompatibleBit : 0, + }; + + Vk api = _context.Api; + if (api.CreateImage(_context.Device, &imageInfo, null, out Image image) != Result.Success) + { + throw new InvalidOperationException("vkCreateImage failed"); + } + + MemoryRequirements requirements = VulkanAllocator.ImageRequirements(_context, image, out bool dedicated); + MemoryAllocation allocation = _context.Allocator.Allocate( + requirements, MemoryPropertyFlags.DeviceLocalBit, linear: false, + $"a {width}x{height} {format} image", poolClass, dedicated, default, image); + if (api.BindImageMemory(_context.Device, image, allocation.Memory, allocation.Offset) != Result.Success) + { + api.DestroyImage(_context.Device, image, null); + _context.Allocator.Free(allocation); + throw new InvalidOperationException("vkBindImageMemory failed"); + } + + uint viewLayers = cube ? 6 : layers; + var viewInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = image, + ViewType = cube ? ImageViewType.TypeCube + : layers > 1 ? ImageViewType.Type2DArray + : ImageViewType.Type2D, + Format = format, + SubresourceRange = new ImageSubresourceRange(aspect, 0, mipLevels, 0, viewLayers), + }; + if (api.CreateImageView(_context.Device, &viewInfo, null, out ImageView view) != Result.Success) + { + api.DestroyImage(_context.Device, image, null); + _context.Allocator.Free(allocation); + throw new InvalidOperationException("vkCreateImageView failed"); + } + + var texture = new VulkanTexture(_context) + { + Image = image, + Allocation = allocation, + View = view, + Format = format, + Width = width, + Height = height, + MipLevels = mipLevels, + Layers = viewLayers, + Cube = cube, + Aspect = aspect, + Usage = usage, + }; + + if (_context.PoisonFreshResources) Poison(texture); + + return Register(texture); + } + + /// + /// Creates a single-level 3D colour texture, for the bindless table's + /// sampler3D placeholder: the game creates no 3D textures, but the + /// array's slot 0 still needs a 3D view. Uploads address it as layer 0 of a + /// one-texel-deep image. + /// + public int CreateVolume(uint width, uint height, uint depth, Format format) + { + width = Math.Max(1, width); + height = Math.Max(1, height); + depth = Math.Max(1, depth); + + var imageInfo = new ImageCreateInfo + { + SType = StructureType.ImageCreateInfo, + ImageType = ImageType.Type3D, + Format = format, + Extent = new Extent3D(width, height, depth), + MipLevels = 1, + ArrayLayers = 1, + Samples = SampleCountFlags.Count1Bit, + Tiling = ImageTiling.Optimal, + Usage = ImageUsageFlags.SampledBit | ImageUsageFlags.TransferDstBit | ImageUsageFlags.TransferSrcBit, + SharingMode = SharingMode.Exclusive, + InitialLayout = ImageLayout.Undefined, + }; + + Vk api = _context.Api; + if (api.CreateImage(_context.Device, &imageInfo, null, out Image image) != Result.Success) + { + throw new InvalidOperationException("vkCreateImage failed for a 3D texture"); + } + + MemoryRequirements requirements = VulkanAllocator.ImageRequirements(_context, image, out bool dedicated); + MemoryAllocation allocation = _context.Allocator.Allocate( + requirements, MemoryPropertyFlags.DeviceLocalBit, linear: false, + $"a {width}x{height}x{depth} {format} image", MemoryPoolClass.DeviceImages, dedicated, default, image); + if (api.BindImageMemory(_context.Device, image, allocation.Memory, allocation.Offset) != Result.Success) + { + api.DestroyImage(_context.Device, image, null); + _context.Allocator.Free(allocation); + throw new InvalidOperationException("vkBindImageMemory failed for a 3D texture"); + } + + var viewInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = image, + ViewType = ImageViewType.Type3D, + Format = format, + SubresourceRange = new ImageSubresourceRange(ImageAspectFlags.ColorBit, 0, 1, 0, 1), + }; + if (api.CreateImageView(_context.Device, &viewInfo, null, out ImageView view) != Result.Success) + { + api.DestroyImage(_context.Device, image, null); + _context.Allocator.Free(allocation); + throw new InvalidOperationException("vkCreateImageView failed for a 3D texture"); + } + + var texture = new VulkanTexture(_context) + { + Image = image, + Allocation = allocation, + View = view, + Format = format, + Width = width, + Height = height, + MipLevels = 1, + Layers = 1, + Volume = true, + Aspect = ImageAspectFlags.ColorBit, + }; + + if (_context.PoisonFreshResources) Poison(texture); + + return Register(texture); + } + + /// + /// Poison mode: fills every level and layer of a new image with + /// 's value for its format, so a read of content + /// nobody wrote is loud instead of whatever the allocator's memory held. + /// Recorded into the upload batch like any upload: a fresh texture has no use + /// yet, so the clear runs before anything that could read it. + /// + private void Poison(VulkanTexture texture) + { + if (VulkanPoison.IsCompressed(texture.Format)) return; + + CommandBuffer commandBuffer = _uploads.BeginRecording(inlineInFrame: false); + try + { + TransitionTexture(commandBuffer, texture, ImageLayout.TransferDstOptimal); + var range = new ImageSubresourceRange(texture.Aspect, 0, texture.MipLevels, 0, texture.Layers); + if (texture.Aspect == ImageAspectFlags.DepthBit) + { + var depth = new ClearDepthStencilValue(VulkanPoison.Depth, 0); + _context.Api.CmdClearDepthStencilImage(commandBuffer, texture.Image, + ImageLayout.TransferDstOptimal, &depth, 1, &range); + } + else + { + ClearColorValue color = VulkanPoison.ColorFor(texture.Format); + _context.Api.CmdClearColorImage(commandBuffer, texture.Image, + ImageLayout.TransferDstOptimal, &color, 1, &range); + } + + // Out of TRANSFER_DST, into the layout an upload leaves a texture in. + // The tracker models the clear and a following copy as the same + // Transfer write, so without this the first upload recorded right + // after the clear gets no barrier; synchronization2 tells CLEAR from + // COPY, and the layer reports WRITE_AFTER_WRITE (write_barriers = 0, + // seq_no 2, 2026-09-15, the first texture of a poisoned device). The + // poison stays: the layout change keeps the contents. + TransitionTexture(commandBuffer, texture, ResourceUsage.SampleFragment); + } + finally + { + _uploads.EndRecording(); + } + } + + /// + /// Uploads pixels into a region: staged, copied, and back to a shader-readable + /// layout, recorded into the upload batch that the next frame submission runs + /// first. Nothing waits. When the frame command buffer being recorded already + /// used the texture, the copy goes inline into it instead, so a draw recorded + /// before the upload still sees the old texels, as on GL. + /// + public void Upload( + int textureId, int level, int x, int y, uint width, uint height, + IntPtr pixels, int bytesPerPixel, uint layer = 0) + { + VulkanTexture? texture = Get(textureId); + if (texture == null || pixels == IntPtr.Zero) return; + + ulong size = (ulong)width * height * (ulong)bytesPerPixel; + if (size == 0) return; + + VulkanStats.NoteUploadRequest(); + CommandBuffer commandBuffer = _uploads.BeginRecording(_uploads.UsedByPendingFrame(texture.FrameUse)); + try + { + // Again under the lock: a delete on another thread either came first + // (nothing to upload to) or retires the texture against this batch's + // Transfer value, so the batch never names a destroyed image. + if (!ReferenceEquals(Get(textureId), texture)) return; + + StagingSlice staging = _uploads.Stage(size); + System.Buffer.MemoryCopy((void*)pixels, (void*)staging.Pointer, (long)size, (long)size); + + if (_context.CheckpointsAvailable) + { + _context.CmdSetCheckpoint(commandBuffer, CheckpointMarker.Upload(textureId, width, height)); + } + + TransitionTexture(commandBuffer, texture, ImageLayout.TransferDstOptimal); + + var region = new BufferImageCopy + { + BufferOffset = staging.Offset, + ImageSubresource = new ImageSubresourceLayers(texture.Aspect, (uint)level, layer, 1), + ImageOffset = new Offset3D(x, y, 0), + ImageExtent = new Extent3D(width, height, 1), + }; + _context.Api.CmdCopyBufferToImage(commandBuffer, staging.Buffer, texture.Image, + ImageLayout.TransferDstOptimal, 1, ®ion); + + TransitionTexture(commandBuffer, texture, ImageLayout.ShaderReadOnlyOptimal); + } + finally + { + _uploads.EndRecording(); + } + } + + /// + /// Builds the mip chain by successive blits, which is how every Vulkan + /// implementation of glGenerateMipmap works. Batched or inline by the same + /// rule as , so it follows the uploads it is built from. + /// + public void GenerateMipmaps(int textureId) + { + VulkanTexture? texture = Get(textureId); + if (texture == null || texture.MipLevels <= 1) return; + + VulkanStats.NoteUploadRequest(); + CommandBuffer commandBuffer = _uploads.BeginRecording(_uploads.UsedByPendingFrame(texture.FrameUse)); + try + { + if (!ReferenceEquals(Get(textureId), texture)) return; + + Vk api = _context.Api; + int mipWidth = (int)texture.Width; + int mipHeight = (int)texture.Height; + + if (_context.CheckpointsAvailable) + { + _context.CmdSetCheckpoint(commandBuffer, CheckpointMarker.Mipmaps(textureId, texture.MipLevels)); + } + + TransitionTexture(commandBuffer, texture, ResourceUsage.TransferSrc); + + for (uint level = 1; level < texture.MipLevels; level++) + { + int nextWidth = Math.Max(1, mipWidth / 2); + int nextHeight = Math.Max(1, mipHeight / 2); + + // The level is about to be overwritten whole: discard it. + TransitionRange(commandBuffer, texture, level, 1, ResourceUsage.TransferDst, discard: true); + + var blit = new ImageBlit + { + SrcSubresource = new ImageSubresourceLayers(texture.Aspect, level - 1, 0, texture.Layers), + DstSubresource = new ImageSubresourceLayers(texture.Aspect, level, 0, texture.Layers), + }; + blit.SrcOffsets.Element0 = new Offset3D(0, 0, 0); + blit.SrcOffsets.Element1 = new Offset3D(mipWidth, mipHeight, 1); + blit.DstOffsets.Element0 = new Offset3D(0, 0, 0); + blit.DstOffsets.Element1 = new Offset3D(nextWidth, nextHeight, 1); + + api.CmdBlitImage(commandBuffer, + texture.Image, ImageLayout.TransferSrcOptimal, + texture.Image, ImageLayout.TransferDstOptimal, + 1, &blit, Filter.Linear); + + TransitionRange(commandBuffer, texture, level, 1, ResourceUsage.TransferSrc, discard: false); + + mipWidth = nextWidth; + mipHeight = nextHeight; + } + + // Every level is TRANSFER_SRC again, so the tracker merged the + // per-level entries back into one and this is a single barrier. + TransitionTexture(commandBuffer, texture, ResourceUsage.SampleFragment); + } + finally + { + _uploads.EndRecording(); + } + } + + /// + /// Applies a glTexParameter. Nothing touches the GPU: the state lives on the + /// texture and resolves to a cached sampler when it is next bound. + /// + public void SetParameter(int textureId, int parameterName, float value) + { + VulkanTexture? texture = Get(textureId); + if (texture == null) return; + + SamplerState state = texture.State; + int integer = (int)value; + + texture.State = parameterName switch + { + GlEnums.TextureMinFilter => ApplyMinFilter(state, integer), + GlEnums.TextureMagFilter => state with { MagFilter = GlEnums.FilterFrom(integer) }, + GlEnums.TextureWrapS => state with { AddressU = GlEnums.AddressModeFrom(integer) }, + GlEnums.TextureWrapT => state with { AddressV = GlEnums.AddressModeFrom(integer) }, + GlEnums.TextureLodBias => state with { LodBias = value }, + GlEnums.TextureMaxLevel => state with { MaxLevel = integer }, + GlEnums.TextureCompareMode => state with + { + CompareEnable = integer == GlEnums.TextureCompareRefToTexture, + }, + _ => state, + }; + } + + private static SamplerState ApplyMinFilter(SamplerState state, int glFilter) + { + (Filter filter, SamplerMipmapMode mode) = GlEnums.MinFilterFrom(glFilter); + return state with + { + MinFilter = filter, + MipmapMode = mode, + Mipmapped = GlEnums.MinFilterUsesMipmaps(glFilter), + }; + } + + /// + /// Vulkan offers four fixed border colours where GL takes an arbitrary one. + /// The SSAO targets use opaque white; anything else rounds to the nearest of + /// the four rather than failing. + /// + public void SetBorderColor(int textureId, float r, float g, float b, float a) + { + VulkanTexture? texture = Get(textureId); + if (texture == null) return; + + bool opaque = a >= 0.5f; + bool white = (r + g + b) / 3f >= 0.5f; + + texture.State = texture.State with + { + BorderColor = opaque + ? white ? BorderColor.FloatOpaqueWhite : BorderColor.FloatOpaqueBlack + : BorderColor.FloatTransparentBlack, + }; + } + + /// + /// Called from with the physical texture that id named, on + /// whichever thread deleted it, under the upload lock. The bindless table + /// retires the texture's slots here. + /// + public Action? Deleted { get; set; } + + public void Delete(int textureId, FrameRing? ring = null) + { + // Under the upload lock; see Upload. Retiring inside it keys the entry on + // the Transfer value of any batch that recorded this texture already. + _uploads.EnterLock(); + try + { + // A texture served by a transient image this frame deletes its own image. + RestoreBindingLocked(textureId); + VulkanTexture? texture = Get(textureId); + if (texture == null) return; + + _textures[textureId] = null; + _freeIds.Push(textureId); + + // Marked before the hook: a lookup that took the texture before this + // delete and reaches the bindless table after Release must not allocate + // a slot nothing would ever retire. + texture.Released = true; + + // Before the texture is retired, under the same timeline values: its + // bindless slots then outlive every frame that could sample them. + Deleted?.Invoke(texture); + + // Handing it to the ring means it outlives any frame still referencing it. + if (ring != null) ring.DeferDeletion(texture); + else texture.Dispose(); + } + finally + { + _uploads.ExitLock(); + } + } + + // ------------------------------------------------------------ transient binds + + // Texture ids that resolve to a transient image for the current frame, and the + // texture each owns. See TransientAllocator.Bind. + private readonly Dictionary _reboundOriginals = new(); + + /// + /// Makes resolve to 's image until + /// . Every path that looks the id up (attachments, + /// samplers, readback) then uses that image, which is how an aliased transient + /// takes the place of a client texture for one frame. + /// + public void Rebind(int id, int physicalId) + { + _uploads.EnterLock(); + try + { + VulkanTexture? physical = Get(physicalId); + if (physical == null || id <= 0 || id >= _textures.Count || id == physicalId) return; + if (!_reboundOriginals.ContainsKey(id)) _reboundOriginals.Add(id, _textures[id]); + _textures[id] = physical; + } + finally + { + _uploads.ExitLock(); + } + } + + /// Undoes every . + public void RestoreBindings() + { + if (_reboundOriginals.Count == 0) return; + _uploads.EnterLock(); + try + { + foreach (KeyValuePair entry in _reboundOriginals) _textures[entry.Key] = entry.Value; + _reboundOriginals.Clear(); + } + finally + { + _uploads.ExitLock(); + } + } + + /// Undoes a of one id; nothing when it is not rebound. + public void RestoreBinding(int id) + { + if (_reboundOriginals.Count == 0) return; + _uploads.EnterLock(); + try + { + RestoreBindingLocked(id); + } + finally + { + _uploads.ExitLock(); + } + } + + private void RestoreBindingLocked(int id) + { + if (_reboundOriginals.Remove(id, out VulkanTexture? original)) _textures[id] = original; + } + + /// Whether currently resolves to another texture's image. + public bool IsRebound(int id) => _reboundOriginals.ContainsKey(id); + + /// The texture's contents stop mattering: its next use transitions from UNDEFINED. + public void DiscardContents(VulkanTexture texture) + { + ResourceStateTracker tracker = texture.Sync; + lock (tracker) tracker.Discard(); + } + + // ------------------------------------------------------------------ barriers + + // Barriers for the immediate transitions below. Uploads record from any + // thread under the upload lock and the frame thread records without it, so + // the shared batcher is used under its own lock, one Require+Flush at a time. + private readonly BarrierBatcher _barriers; + private readonly object _barrierLock = new(); + + /// + /// Whether a rendering scope is open in a command buffer; the device answers + /// for its frame command buffer. Every batcher from + /// asks it at flush time (debug builds reject a flush inside a scope). + /// + public Func? ScopeOpen { get; set; } + + /// A batcher for one recording thread (a render target manager, the present path). + public BarrierBatcher CreateBatcher() => + new(_context.Api) { ScopeOpen = commandBuffer => ScopeOpen?.Invoke(commandBuffer) == true }; + + /// + /// Adds a whole-texture use to without flushing, so + /// several textures move in one barrier command. The caller flushes before + /// recording the commands that use them. + /// + public void Require(BarrierBatcher batcher, CommandBuffer commandBuffer, VulkanTexture texture, ResourceUsage usage) + { + _uploads.NoteUse(commandBuffer, texture); + batcher.Require(texture, 0, texture.MipLevels, 0, texture.Layers, usage); + } + + /// + /// for + /// a range of mip levels (every layer): a compute pass reading level n and writing + /// level n + 1 of one image moves each level into its own layout. + /// + public void Require(BarrierBatcher batcher, CommandBuffer commandBuffer, VulkanTexture texture, + uint baseMip, uint mipCount, ResourceUsage usage) + { + _uploads.NoteUse(commandBuffer, texture); + batcher.Require(texture, baseMip, mipCount, 0, texture.Layers, usage); + } + + /// A transition named by layout (tests, a readback restoring what it found); see . + public void TransitionTexture(CommandBuffer commandBuffer, VulkanTexture texture, ImageLayout target) => + TransitionTexture(commandBuffer, texture, UsageState.ForLayout(target)); + + public void TransitionTexture(CommandBuffer commandBuffer, VulkanTexture texture, ResourceUsage usage, + bool discard = false) + { + // Every path that records a texture into a command buffer goes through + // here or Require (attachments, reads, copies, blits), even when no barrier is due. + _uploads.NoteUse(commandBuffer, texture); + TransitionRange(commandBuffer, texture, 0, texture.MipLevels, usage, discard); + } + + private void TransitionRange(CommandBuffer commandBuffer, VulkanTexture texture, + uint baseMip, uint mipCount, ResourceUsage usage, bool discard) + { + lock (_barrierLock) + { + _barriers.Require(texture, baseMip, mipCount, 0, texture.Layers, usage, discard); + _barriers.Flush(commandBuffer); + } + } + + // -------------------------------------------------------------------- helpers + + public static uint MipLevelsFor(uint width, uint height) => + (uint)Math.Floor(Math.Log2(Math.Max(width, height))) + 1; + + public static bool IsDepthFormat(Format format) => format is + Format.D16Unorm or Format.D32Sfloat or Format.D24UnormS8Uint or Format.D32SfloatS8Uint + or Format.X8D24UnormPack32 or Format.D16UnormS8Uint; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + // A rebound id holds a transient image that has its own entry; dispose owners only. + foreach (KeyValuePair entry in _reboundOriginals) _textures[entry.Key] = entry.Value; + _reboundOriginals.Clear(); + foreach (VulkanTexture? texture in _textures) texture?.Dispose(); + _textures.Clear(); + Samplers.Dispose(); + } +} diff --git a/Optimum.Render.Vulkan/Core/VertexLayout.cs b/Optimum.Render.Vulkan/Core/VertexLayout.cs new file mode 100644 index 00000000..67d5d0f4 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VertexLayout.cs @@ -0,0 +1,290 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// One vertex buffer feeding the pipeline. +internal readonly record struct VertexBinding(uint Binding, uint Stride, bool PerInstance); + +/// One vertex attribute read out of a binding. +internal readonly record struct VertexAttribute(uint Location, uint Binding, Format Format, uint Offset); + +/// +/// The vertex input state of a mesh. +/// +/// Vintage Story gives every attribute its own buffer rather than interleaving +/// one - separate allocations for positions, normals, UVs, colours, flags, plus +/// up to four custom parts that are interleaved and may be per-instance. +/// So a binding here is usually one buffer with one attribute, and the custom +/// parts are the exception. +/// +/// There are on the order of fifteen distinct layouts in the whole game, so they +/// are interned and the id goes into the pipeline key. +/// +internal sealed class VertexLayoutDescription : IEquatable +{ + public VertexBinding[] Bindings { get; } + public VertexAttribute[] Attributes { get; } + + private readonly int _hash; + + public VertexLayoutDescription(VertexBinding[] bindings, VertexAttribute[] attributes) + { + Bindings = bindings; + Attributes = attributes; + + var hash = new HashCode(); + foreach (VertexBinding binding in bindings) hash.Add(binding); + foreach (VertexAttribute attribute in attributes) hash.Add(attribute); + _hash = hash.ToHashCode(); + } + + /// The layout of a pass that generates its vertices in the shader. + public static VertexLayoutDescription Empty { get; } = + new(Array.Empty(), Array.Empty()); + + /// + /// The binding the constant-default attribute buffer occupies. + /// + /// Meshes number their bindings from zero and never approach this, and the + /// Vulkan minimum for maxVertexInputBindings is 16, so 15 is always available + /// and never collides. + /// + public const uint DefaultAttributeBinding = 15; + + /// + /// Adds constant-default attributes for every location the program declares + /// but this layout does not provide. + /// + /// GL answers a read of an unbound vertex attribute with the current generic + /// attribute, which defaults to (0, 0, 0, 1); Vulkan leaves it undefined. + /// That difference is not cosmetic - gui.fsh discards a fragment based on a + /// damage effect fed by an attribute the GUI quad never carries, so undefined + /// there means the entire interface vanishes with no validation message. + /// + public VertexLayoutDescription WithDefaultsFor(IReadOnlyList declared) + { + List? added = null; + foreach (VertexInputSlot slot in declared) + { + bool present = false; + foreach (VertexAttribute attribute in Attributes) + { + if (attribute.Location == (uint)slot.Location) { present = true; break; } + } + if (present) continue; + + added ??= new List(); + added.Add(new VertexAttribute( + (uint)slot.Location, DefaultAttributeBinding, + DefaultFormatFor(slot.Type), DefaultOffsetFor(slot.Type))); + } + + if (added == null) return this; + + var bindings = new VertexBinding[Bindings.Length + 1]; + Array.Copy(Bindings, bindings, Bindings.Length); + // Stride zero: every vertex reads the same constant. + bindings[^1] = new VertexBinding(DefaultAttributeBinding, 0, PerInstance: false); + + var attributes = new VertexAttribute[Attributes.Length + added.Count]; + Array.Copy(Attributes, attributes, Attributes.Length); + for (int i = 0; i < added.Count; i++) attributes[Attributes.Length + i] = added[i]; + + return new VertexLayoutDescription(bindings, attributes); + } + + /// + /// Integer attributes have to read integer zeros and floats floating-point + /// ones, so the default buffer holds both and the type picks the half. + /// + private static uint DefaultOffsetFor(GlslType type) => IsIntegerType(type) ? 16u : 0u; + + private static Format DefaultFormatFor(GlslType type) + { + bool integer = IsIntegerType(type); + bool unsigned = IsUnsignedType(type); + return type.ComponentCount switch + { + 1 => integer ? (unsigned ? Format.R32Uint : Format.R32Sint) : Format.R32Sfloat, + 2 => integer ? (unsigned ? Format.R32G32Uint : Format.R32G32Sint) : Format.R32G32Sfloat, + 3 => integer ? (unsigned ? Format.R32G32B32Uint : Format.R32G32B32Sint) : Format.R32G32B32Sfloat, + _ => integer ? (unsigned ? Format.R32G32B32A32Uint : Format.R32G32B32A32Sint) : Format.R32G32B32A32Sfloat, + }; + } + + private static bool IsUnsignedType(GlslType type) => + type.Name.StartsWith("u", StringComparison.Ordinal) || type.Name == "uint"; + + private static bool IsIntegerType(GlslType type) => + type.Name.StartsWith("i", StringComparison.Ordinal) || + type.Name.StartsWith("u", StringComparison.Ordinal) || + type.Name == "int" || type.Name == "uint" || type.Name == "bool"; + + public bool Equals(VertexLayoutDescription? other) + { + if (other is null || other._hash != _hash) return false; + if (Bindings.Length != other.Bindings.Length) return false; + if (Attributes.Length != other.Attributes.Length) return false; + + for (int i = 0; i < Bindings.Length; i++) + { + if (!Bindings[i].Equals(other.Bindings[i])) return false; + } + for (int i = 0; i < Attributes.Length; i++) + { + if (!Attributes[i].Equals(other.Attributes[i])) return false; + } + return true; + } + + public override bool Equals(object? obj) => Equals(obj as VertexLayoutDescription); + public override int GetHashCode() => _hash; +} + +/// +/// Builds a vertex layout the way the game's mesh allocator describes one. +/// +/// The GL path calls glVertexAttribPointer per attribute with a raw type +/// constant, so the mapping from those constants to Vulkan formats is the whole +/// job. Getting one wrong shows up as garbled geometry rather than an error, +/// which is why the mapping is tested rather than trusted. +/// +internal sealed class VertexLayoutBuilder +{ + private readonly List _bindings = new(); + private readonly List _attributes = new(); + private uint _nextLocation; + + /// + /// Adds an attribute backed by its own tightly packed buffer, which is how + /// positions, normals, UVs, colours and flags arrive. + /// + public VertexLayoutBuilder AddDedicated(Format format, uint stride, bool perInstance = false) + { + uint binding = (uint)_bindings.Count; + _bindings.Add(new VertexBinding(binding, stride, perInstance)); + _attributes.Add(new VertexAttribute(_nextLocation++, binding, format, 0)); + return this; + } + + /// + /// Adds an interleaved group sharing one buffer, which is how the custom + /// mesh data parts arrive - several attributes at different offsets within a + /// common stride, optionally advancing per instance. + /// + public VertexLayoutBuilder AddInterleaved( + ReadOnlySpan<(Format Format, uint Offset)> members, uint stride, bool perInstance) + { + uint binding = (uint)_bindings.Count; + _bindings.Add(new VertexBinding(binding, stride, perInstance)); + foreach ((Format format, uint offset) in members) + { + _attributes.Add(new VertexAttribute(_nextLocation++, binding, format, offset)); + } + return this; + } + + public VertexLayoutDescription Build() => new(_bindings.ToArray(), _attributes.ToArray()); + + // ------------------------------------------------------------ format mapping + + private const int GlByte = 0x1400; + private const int GlUnsignedByte = 0x1401; + private const int GlShort = 0x1402; + private const int GlUnsignedShort = 0x1403; + private const int GlInt = 0x1404; + private const int GlUnsignedInt = 0x1405; + private const int GlFloat = 0x1406; + private const int GlInt2101010Rev = 0x8D9F; + + /// + /// Maps a glVertexAttribPointer description to a Vulkan format. + /// + /// 1 to 4, or 4 for a packed 2-10-10-10 normal. + /// The GL type constant. + /// + /// True when GL would scale integers into [0,1] or [-1,1]. False with + /// false means the shader sees the raw value as a + /// float; true with integer means an integer attribute. + /// + /// True for glVertexAttribIPointer. + public static Format FormatFor(int components, int glType, bool normalized, bool integer) + { + // The packed normal format the game uses for entity and particle normals. + if (glType == GlInt2101010Rev) + { + return normalized ? Format.A2B10G10R10SNormPack32 : Format.A2B10G10R10SintPack32; + } + + return glType switch + { + GlFloat => components switch + { + 1 => Format.R32Sfloat, + 2 => Format.R32G32Sfloat, + 3 => Format.R32G32B32Sfloat, + _ => Format.R32G32B32A32Sfloat, + }, + GlUnsignedByte => Select(components, normalized, integer, + (Format.R8Unorm, Format.R8G8Unorm, Format.R8G8B8Unorm, Format.R8G8B8A8Unorm), + (Format.R8Uint, Format.R8G8Uint, Format.R8G8B8Uint, Format.R8G8B8A8Uint), + (Format.R8Uscaled, Format.R8G8Uscaled, Format.R8G8B8Uscaled, Format.R8G8B8A8Uscaled)), + GlByte => Select(components, normalized, integer, + (Format.R8SNorm, Format.R8G8SNorm, Format.R8G8B8SNorm, Format.R8G8B8A8SNorm), + (Format.R8Sint, Format.R8G8Sint, Format.R8G8B8Sint, Format.R8G8B8A8Sint), + (Format.R8Sscaled, Format.R8G8Sscaled, Format.R8G8B8Sscaled, Format.R8G8B8A8Sscaled)), + GlUnsignedShort => Select(components, normalized, integer, + (Format.R16Unorm, Format.R16G16Unorm, Format.R16G16B16Unorm, Format.R16G16B16A16Unorm), + (Format.R16Uint, Format.R16G16Uint, Format.R16G16B16Uint, Format.R16G16B16A16Uint), + (Format.R16Uscaled, Format.R16G16Uscaled, Format.R16G16B16Uscaled, Format.R16G16B16A16Uscaled)), + GlShort => Select(components, normalized, integer, + (Format.R16SNorm, Format.R16G16SNorm, Format.R16G16B16SNorm, Format.R16G16B16A16SNorm), + (Format.R16Sint, Format.R16G16Sint, Format.R16G16B16Sint, Format.R16G16B16A16Sint), + (Format.R16Sscaled, Format.R16G16Sscaled, Format.R16G16B16Sscaled, Format.R16G16B16A16Sscaled)), + GlUnsignedInt => components switch + { + 1 => Format.R32Uint, + 2 => Format.R32G32Uint, + 3 => Format.R32G32B32Uint, + _ => Format.R32G32B32A32Uint, + }, + GlInt => components switch + { + 1 => Format.R32Sint, + 2 => Format.R32G32Sint, + 3 => Format.R32G32B32Sint, + _ => Format.R32G32B32A32Sint, + }, + _ => Format.R32G32B32A32Sfloat, + }; + } + + private static Format Select( + int components, bool normalized, bool integer, + (Format, Format, Format, Format) norm, + (Format, Format, Format, Format) asInteger, + (Format, Format, Format, Format) scaled) + { + (Format one, Format two, Format three, Format four) = + integer ? asInteger : normalized ? norm : scaled; + + return components switch { 1 => one, 2 => two, 3 => three, _ => four }; + } + + /// Bytes one vertex of this format occupies. + public static uint SizeOf(int components, int glType) + { + if (glType == GlInt2101010Rev) return 4; + + uint componentSize = glType switch + { + GlByte or GlUnsignedByte => 1u, + GlShort or GlUnsignedShort => 2u, + _ => 4u, + }; + return componentSize * (uint)Math.Clamp(components, 1, 4); + } +} diff --git a/Optimum.Render.Vulkan/Core/VulkanAllocator.cs b/Optimum.Render.Vulkan/Core/VulkanAllocator.cs new file mode 100644 index 00000000..feec44ad --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VulkanAllocator.cs @@ -0,0 +1,916 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Text; +using Silk.NET.Vulkan; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// What a piece of memory is for. Each class pools separately with its own block +/// size, so a long-lived static mesh never shares a block with per-frame data and +/// the cap on ReBAR can be enforced by class rather than guessed from flags. +/// +internal enum MemoryPoolClass +{ + /// Sampled textures and attachments: optimally tiled images, 128 MiB blocks. + DeviceImages = 0, + + /// Long-lived buffers (static meshes device-local, dynamic meshes host-visible): 64 MiB blocks. + DeviceBuffers = 1, + + /// Host-side transfer memory (staging, readback arenas, a ReBAR miss's fall-through): 32 MiB blocks. + Staging = 2, + + /// + /// Device-local and host-visible memory for per-frame dynamic data only + /// (uniform ring, indirect ring): 16 MiB blocks, capped at + /// min(192 MiB, budget x 0.25). A miss falls through to . + /// + ReBar = 3, + + /// Frame-graph transient attachments (reserved for Phase 2): 64 MiB blocks. + Transient = 4, + + /// + /// One allocation per resource: the driver asked for it + /// (VkMemoryDedicatedRequirements) or the resource is at least a quarter of + /// its class's block size. + /// + Dedicated = 5, +} + +/// A region of a memory block handed to one resource. +internal readonly struct MemoryAllocation +{ + public DeviceMemory Memory { get; init; } + public ulong Offset { get; init; } + public ulong Size { get; init; } + + /// Host pointer to this region, or zero when the memory is not mapped. + public IntPtr Mapped { get; init; } + + internal MemoryBlock? Block { get; init; } + + public bool IsValid => Block != null; +} + +/// +/// One vkAllocateMemory, divided up among many resources. +/// +/// Free space is tracked as ranges in address order, so neighbouring frees merge +/// back into one range and the block does not fragment into confetti as chunk +/// meshes come and go. +/// +internal sealed unsafe class MemoryBlock : IDisposable +{ + private readonly struct FreeRange + { + public FreeRange(ulong offset, ulong size) + { + Offset = offset; + Size = size; + } + + public ulong Offset { get; } + public ulong Size { get; } + public ulong End => Offset + Size; + } + + private readonly VulkanContext _context; + private readonly List _free = new(); + private bool _disposed; + + public DeviceMemory Memory { get; } + public ulong Size { get; } + public uint TypeIndex { get; } + + /// The heap the block's memory type draws from. + public uint HeapIndex { get; } + + /// + /// The pool class the block belongs to. A dedicated block keeps the class of + /// the resource it backs, so a dedicated ReBAR resource still counts against + /// the ReBAR cap. + /// + public MemoryPoolClass Class { get; } + + /// + /// Whether this block holds linear resources (buffers) or optimally tiled + /// ones (images). They are never mixed, which is what makes + /// bufferImageGranularity irrelevant here: the spec only requires padding + /// between the two kinds, and there is never a boundary between them. + /// + public bool Linear { get; } + + /// Set when the block backs exactly one resource. + public bool Dedicated { get; } + + /// Base host pointer when the memory type is host visible. + public IntPtr Mapped { get; private set; } + + public ulong Used { get; private set; } + + public bool IsEmpty => Used == 0; + + /// + /// The allocator frame at which a pooled block last became empty, or -1 while + /// it holds anything. Empty blocks are freed after + /// frames. + /// + internal long EmptySinceFrame = -1; + + public MemoryBlock( + VulkanContext context, ulong size, uint typeIndex, bool linear, bool dedicated, bool hostVisible) + : this(context, size, typeIndex, 0, MemoryPoolClass.DeviceBuffers, linear, dedicated, hostVisible, + default, default) + { + } + + public MemoryBlock( + VulkanContext context, ulong size, uint typeIndex, uint heapIndex, MemoryPoolClass poolClass, + bool linear, bool dedicated, bool hostVisible, Buffer dedicatedBuffer, Image dedicatedImage) + { + _context = context; + Size = size; + TypeIndex = typeIndex; + HeapIndex = heapIndex; + Class = poolClass; + Linear = linear; + Dedicated = dedicated; + + // A dedicated block names its resource, which lets the driver place it + // (and is mandatory when the resource reported requiresDedicatedAllocation). + var dedicatedInfo = new MemoryDedicatedAllocateInfo + { + SType = StructureType.MemoryDedicatedAllocateInfo, + Buffer = dedicatedBuffer, + Image = dedicatedImage, + }; + bool namesResource = dedicated && (dedicatedBuffer.Handle != 0 || dedicatedImage.Handle != 0); + + var allocateInfo = new MemoryAllocateInfo + { + SType = StructureType.MemoryAllocateInfo, + PNext = namesResource ? &dedicatedInfo : null, + AllocationSize = size, + MemoryTypeIndex = typeIndex, + }; + + Memory = VulkanMemory.Allocate(context, allocateInfo, + $"a {size} byte {(dedicated ? "dedicated" : "pooled")} {poolClass} memory block"); + + if (hostVisible) + { + void* mapped; + // Mapped once for the block's whole life. Mapping is not free and a + // resource may be written from any thread, so per-resource mapping + // would be both slower and harder to synchronise. + if (context.Api.MapMemory(context.Device, Memory, 0, size, 0, &mapped) == Result.Success) + { + Mapped = (IntPtr)mapped; + } + } + + _free.Add(new FreeRange(0, size)); + } + + public bool TryAllocate(ulong size, ulong alignment, out ulong offset) + { + offset = 0; + if (size == 0 || _disposed) return false; + + for (int i = 0; i < _free.Count; i++) + { + FreeRange range = _free[i]; + + ulong aligned = alignment <= 1 + ? range.Offset + : (range.Offset + alignment - 1) / alignment * alignment; + + ulong padding = aligned - range.Offset; + if (range.Size < padding || range.Size - padding < size) continue; + + ulong tail = range.Size - padding - size; + + // The alignment padding stays free rather than being lost, so a + // later smaller or less strictly aligned resource can use it. + _free.RemoveAt(i); + if (tail > 0) _free.Insert(i, new FreeRange(aligned + size, tail)); + if (padding > 0) _free.Insert(i, new FreeRange(range.Offset, padding)); + + Used += size; + offset = aligned; + return true; + } + + return false; + } + + public void Free(ulong offset, ulong size) + { + if (_disposed || size == 0) return; + + Used -= Math.Min(Used, size); + + int index = 0; + while (index < _free.Count && _free[index].Offset < offset) index++; + + ulong start = offset; + ulong end = offset + size; + + // Merge with the range before, if they touch. + if (index > 0 && _free[index - 1].End == start) + { + start = _free[index - 1].Offset; + _free.RemoveAt(index - 1); + index--; + } + + // And with the range after. + if (index < _free.Count && _free[index].Offset == end) + { + end = _free[index].End; + _free.RemoveAt(index); + } + + _free.Insert(index, new FreeRange(start, end - start)); + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + if (Mapped != IntPtr.Zero) + { + _context.Api.UnmapMemory(_context.Device, Memory); + Mapped = IntPtr.Zero; + } + + _context.Api.FreeMemory(_context.Device, Memory, null); + VulkanMemory.NoteFree(); + _free.Clear(); + } +} + +/// The allocator's state at one moment, for the stats.memory line and tests. +internal readonly record struct MemorySnapshot( + int Blocks, + int DedicatedBlocks, + ulong ReBarUsed, + ulong ReBarCap, + long ReBarMisses, + long EmptyBlocksFreed, + bool BudgetExtension, + ulong[] ClassBytes, + ulong[] HeapUsed, + ulong[] HeapBudget); + +/// +/// Hands resources memory out of a few large blocks instead of giving each its +/// own allocation. +/// +/// A device allocation is not a cheap object. The driver tracks every one of +/// them and builds a residency list over the whole set on each submit, so cost +/// grows with the count rather than with the bytes. Backing every buffer and +/// image individually put a loaded world at eighteen thousand live allocations, +/// where frames took 200 ms; the same world's memory in a few dozen blocks is +/// the difference between four frames a second and fifty. The allocation limit +/// the spec exposes - commonly 4096 - is the same problem stated as a hard cap, +/// and NVIDIA not enforcing one is why this degraded instead of failing. +/// +/// Phase 1B step 5: memory is pooled per and +/// memory type. Buffers and images are kept in separate blocks so +/// bufferImageGranularity never applies; anything the driver wants dedicated, or +/// large enough to waste a quarter of its class's block, gets its own. Heap +/// budgets come from VK_EXT_memory_budget when the device has it (heap x 0.7 +/// otherwise). Empty blocks are freed after +/// frames, or at once when a heap is over its budget. +/// +internal sealed unsafe class VulkanAllocator : IDisposable +{ + public const int PoolClassCount = 6; + + private const ulong MiB = 1024UL * 1024; + + /// Frames a pooled block stays empty before it is freed. + public const int EmptyBlockFrames = 120; + + /// The ReBAR cap's ceiling; the cap is min(this, budget x 0.25). + public const ulong ReBarCapCeiling = 192 * MiB; + + /// Budget as a share of heap size when VK_EXT_memory_budget is absent. + public const double FallbackBudgetShare = 0.7; + + /// ReBAR misses reported through ; the counter keeps counting past it. + private const int LoggedMissLimit = 32; + + /// + /// Set by OPTIMUM_VULKAN_DEDICATED_MEMORY=1 to give every resource its own + /// vkAllocateMemory, which is what this backend did before pooling existed. + /// + /// It is ruinously slow - that is the whole reason pooling is here - but it + /// removes every question of one resource landing on another's bytes, so a + /// rendering fault that survives it is not a suballocation fault. Keeping + /// the old behaviour reachable is what makes that a one-run experiment + /// rather than a bisect. + /// + private static readonly bool AlwaysDedicated = + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_DEDICATED_MEMORY") == "1"; + + /// OPTIMUM_VULKAN_NO_REBAR=1 forces every ReBAR request down the fall-through path. + private static readonly bool ReBarDisabled = + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_NO_REBAR") == "1"; + + private readonly VulkanContext _context; + private readonly object _gate = new(); + private readonly Dictionary<(MemoryPoolClass Class, uint TypeIndex, bool Linear), List> _pools = new(); + private readonly List _dedicated = new(); + private readonly PhysicalDeviceMemoryProperties _memoryProperties; + private readonly ulong[] _heapUsed; + private readonly ulong[] _heapBudget; + private readonly ulong[] _classBytes = new ulong[PoolClassCount]; + private ulong _reBarUsed; + // Block bytes of the Transient class, dedicated ones included, and their peak since the last take. + private ulong _transientBytes; + private ulong _transientPeak; + private long _reBarMisses; + private long _emptyBlocksFreed; + private long _frame; + private int _emptyBlocks; + private bool _disposed; + + public VulkanAllocator(VulkanContext context) + { + _context = context; + context.Api.GetPhysicalDeviceMemoryProperties(context.PhysicalDevice, out _memoryProperties); + _heapUsed = new ulong[_memoryProperties.MemoryHeapCount]; + _heapBudget = new ulong[_memoryProperties.MemoryHeapCount]; + BudgetExtension = context.MemoryBudgetAvailable; + RefreshBudgetLocked(); + } + + /// + /// Receives a line for each logged event (ReBAR misses). The device points it + /// at the validation mirror; never at GetError, since a miss is not an error. + /// + public Action? Log { get; set; } + + /// Whether heap budgets come from VK_EXT_memory_budget. + public bool BudgetExtension { get; } + + /// Replaces the ReBAR cap. Tests only. + internal ulong? ReBarCapOverrideForTests { get; set; } + + /// Replaces every heap's budget. Tests only. + internal ulong? HeapBudgetOverrideForTests { get; set; } + + /// Blocks currently held, which is the real vkAllocateMemory count. + public int BlockCount + { + get + { + lock (_gate) + { + return BlockCountLocked(); + } + } + } + + private int BlockCountLocked() + { + int count = _dedicated.Count; + foreach (List blocks in _pools.Values) count += blocks.Count; + return count; + } + + public long ReBarMisses + { + get + { + lock (_gate) return _reBarMisses; + } + } + + /// Block bytes of the given class on ReBAR memory types, dedicated ones included. + public ulong ReBarUsed + { + get + { + lock (_gate) return _reBarUsed; + } + } + + public static ulong BlockSizeOf(MemoryPoolClass poolClass) => poolClass switch + { + MemoryPoolClass.DeviceImages => 128 * MiB, + MemoryPoolClass.DeviceBuffers => 64 * MiB, + MemoryPoolClass.Staging => 32 * MiB, + MemoryPoolClass.ReBar => 16 * MiB, + MemoryPoolClass.Transient => 64 * MiB, + _ => 64 * MiB, + }; + + /// + /// The class a request lands in when the caller does not say: images are + /// DeviceImages; a buffer asking for device-local and host-visible memory is + /// ReBar; every other buffer is DeviceBuffers. + /// + public static MemoryPoolClass InferClass(MemoryPropertyFlags properties, bool linear) + { + if (!linear) return MemoryPoolClass.DeviceImages; + const MemoryPropertyFlags reBar = MemoryPropertyFlags.DeviceLocalBit | MemoryPropertyFlags.HostVisibleBit; + return (properties & reBar) == reBar ? MemoryPoolClass.ReBar : MemoryPoolClass.DeviceBuffers; + } + + public MemoryAllocation Allocate( + MemoryRequirements requirements, MemoryPropertyFlags properties, bool linear, string what) => + Allocate(requirements, properties, linear, what, InferClass(properties, linear), false, default, default); + + /// + /// Allocates for one resource. is the + /// resource's VkMemoryDedicatedRequirements (required or preferred), and the + /// buffer or image handle, when given, is named in a dedicated allocation. + /// + public MemoryAllocation Allocate( + MemoryRequirements requirements, MemoryPropertyFlags properties, bool linear, string what, + MemoryPoolClass poolClass, bool requiresDedicated, Buffer buffer, Image image) + { + if (poolClass == MemoryPoolClass.Dedicated) + { + requiresDedicated = true; + poolClass = InferClass(properties, linear); + } + + lock (_gate) + { + ObjectDisposedException.ThrowIf(_disposed, this); + + if (poolClass == MemoryPoolClass.ReBar) + { + return AllocateReBarLocked(requirements, properties, linear, what, requiresDedicated, buffer, image); + } + + uint typeIndex = FindMemoryType(requirements.MemoryTypeBits, properties, Avoided(properties)); + return AllocateLocked(requirements, typeIndex, poolClass, linear, what, requiresDedicated, buffer, image); + } + } + + /// + /// ReBAR holds only per-frame dynamic data and is capped. A request that finds + /// no ReBAR type, or would take the class past its cap, is counted, logged and + /// served from host-visible staging memory instead. + /// + private MemoryAllocation AllocateReBarLocked( + MemoryRequirements requirements, MemoryPropertyFlags properties, bool linear, string what, + bool requiresDedicated, Buffer buffer, Image image) + { + MemoryPropertyFlags wanted = properties | MemoryPropertyFlags.DeviceLocalBit | MemoryPropertyFlags.HostVisibleBit; + string? miss = null; + + if (ReBarDisabled) + { + miss = "OPTIMUM_VULKAN_NO_REBAR=1"; + } + else if (!TryFindMemoryType(requirements.MemoryTypeBits, wanted, 0, out uint typeIndex)) + { + miss = "no device-local host-visible memory type"; + } + else + { + bool dedicated = AlwaysDedicated || requiresDedicated + || requirements.Size >= BlockSizeOf(MemoryPoolClass.ReBar) / 4; + var key = (MemoryPoolClass.ReBar, typeIndex, linear); + + if (!dedicated && _pools.TryGetValue(key, out List? pool)) + { + foreach (MemoryBlock candidate in pool) + { + if (candidate.TryAllocate(requirements.Size, requirements.Alignment, out ulong offset)) + { + NoteFilled(candidate); + return Describe(candidate, offset, requirements.Size); + } + } + } + + ulong growth = dedicated ? requirements.Size : BlockSizeOf(MemoryPoolClass.ReBar); + ulong cap = ReBarCapLocked(typeIndex); + if (_reBarUsed + growth > cap) + { + miss = "cap " + cap + " bytes reached (" + _reBarUsed + " used, " + growth + " more needed)"; + } + else + { + return AllocateLocked(requirements, typeIndex, MemoryPoolClass.ReBar, linear, what, + requiresDedicated, buffer, image); + } + } + + _reBarMisses++; + VulkanStats.NoteRebarFallback(); + if (_reBarMisses <= LoggedMissLimit) + { + string line = "[Optimum] ReBAR miss for " + what + ": " + miss + + "; falling through to host-visible staging memory (miss " + _reBarMisses + ")"; + Log?.Invoke(line); + if (RenderTrace.Enabled) RenderTrace.Write(line); + } + + MemoryPropertyFlags host = (properties & ~MemoryPropertyFlags.DeviceLocalBit) + | MemoryPropertyFlags.HostVisibleBit; + uint hostType = FindMemoryType(requirements.MemoryTypeBits, host, MemoryPropertyFlags.DeviceLocalBit); + return AllocateLocked(requirements, hostType, MemoryPoolClass.Staging, linear, what, requiresDedicated, + buffer, image); + } + + private MemoryAllocation AllocateLocked( + MemoryRequirements requirements, uint typeIndex, MemoryPoolClass poolClass, bool linear, string what, + bool requiresDedicated, Buffer buffer, Image image) + { + MemoryType type = _memoryProperties.MemoryTypes[(int)typeIndex]; + bool hostVisible = (type.PropertyFlags & MemoryPropertyFlags.HostVisibleBit) != 0; + ulong blockSize = BlockSizeOf(poolClass); + + if (AlwaysDedicated || requiresDedicated || requirements.Size >= blockSize / 4) + { + var block = new MemoryBlock( + _context, requirements.Size, typeIndex, type.HeapIndex, poolClass, linear, dedicated: true, + hostVisible, buffer, image); + _dedicated.Add(block); + NoteBlockCreated(block); + + if (!block.TryAllocate(requirements.Size, requirements.Alignment, out ulong dedicatedOffset)) + { + throw new InvalidOperationException("a dedicated block could not satisfy " + what); + } + return Describe(block, dedicatedOffset, requirements.Size); + } + + var key = (poolClass, typeIndex, linear); + if (!_pools.TryGetValue(key, out List? pool)) + { + pool = new List(); + _pools[key] = pool; + } + + foreach (MemoryBlock candidate in pool) + { + if (candidate.TryAllocate(requirements.Size, requirements.Alignment, out ulong offset)) + { + NoteFilled(candidate); + return Describe(candidate, offset, requirements.Size); + } + } + + var fresh = new MemoryBlock( + _context, blockSize, typeIndex, type.HeapIndex, poolClass, linear, dedicated: false, hostVisible, + default, default); + pool.Add(fresh); + NoteBlockCreated(fresh); + + if (!fresh.TryAllocate(requirements.Size, requirements.Alignment, out ulong freshOffset)) + { + throw new InvalidOperationException("a fresh block could not satisfy " + what); + } + return Describe(fresh, freshOffset, requirements.Size); + } + + private void NoteBlockCreated(MemoryBlock block) + { + _heapUsed[block.HeapIndex] += block.Size; + _classBytes[(int)(block.Dedicated ? MemoryPoolClass.Dedicated : block.Class)] += block.Size; + if (block.Class == MemoryPoolClass.ReBar) _reBarUsed += block.Size; + if (block.Class == MemoryPoolClass.Transient) + { + _transientBytes += block.Size; + if (_transientBytes > _transientPeak) _transientPeak = _transientBytes; + } + } + + /// Block bytes of the Transient pool class, dedicated blocks included. + public ulong TransientHeapBytes + { + get + { + lock (_gate) return _transientBytes; + } + } + + /// + /// The Transient class's peak block bytes since the previous call (the stats + /// sample's heap_peak_mib); the next peak starts from the current use. + /// + public ulong TakeTransientHeapPeak() + { + lock (_gate) + { + ulong peak = Math.Max(_transientPeak, _transientBytes); + _transientPeak = _transientBytes; + return peak; + } + } + + private void NoteBlockReleased(MemoryBlock block) + { + _heapUsed[block.HeapIndex] -= Math.Min(_heapUsed[block.HeapIndex], block.Size); + int index = (int)(block.Dedicated ? MemoryPoolClass.Dedicated : block.Class); + _classBytes[index] -= Math.Min(_classBytes[index], block.Size); + if (block.Class == MemoryPoolClass.ReBar) _reBarUsed -= Math.Min(_reBarUsed, block.Size); + if (block.Class == MemoryPoolClass.Transient) _transientBytes -= Math.Min(_transientBytes, block.Size); + } + + private void NoteFilled(MemoryBlock block) + { + if (block.EmptySinceFrame < 0) return; + block.EmptySinceFrame = -1; + _emptyBlocks--; + } + + private static MemoryAllocation Describe(MemoryBlock block, ulong offset, ulong size) => + new() + { + Memory = block.Memory, + Offset = offset, + Size = size, + Mapped = block.Mapped == IntPtr.Zero ? IntPtr.Zero : block.Mapped + (int)offset, + Block = block, + }; + + public void Free(in MemoryAllocation allocation) + { + MemoryBlock? block = allocation.Block; + if (block == null) return; + + lock (_gate) + { + if (_disposed) return; + + block.Free(allocation.Offset, allocation.Size); + + if (block.Dedicated) + { + _dedicated.Remove(block); + NoteBlockReleased(block); + block.Dispose(); + return; + } + + // An emptied block is kept for EmptyBlockFrames frames, so a pool that + // is repeatedly drained and refilled - which chunk streaming does - is + // not paying for an allocation each time. + if (!block.IsEmpty || block.EmptySinceFrame >= 0) return; + block.EmptySinceFrame = _frame; + _emptyBlocks++; + } + } + + /// + /// One frame boundary (the frame ring calls it at BeginFrame): refreshes the + /// heap budgets now and then, and frees pooled blocks that stayed empty for + /// frames, or every empty block at once while a + /// heap is over its budget. + /// + public void AdvanceFrame() + { + lock (_gate) + { + if (_disposed) return; + _frame++; + + if (_frame % 60 == 0) RefreshBudgetLocked(); + if (_emptyBlocks == 0) return; + + bool pressure = false; + for (int heap = 0; heap < _heapUsed.Length; heap++) + { + if (_heapUsed[heap] > HeapBudgetLocked(heap)) pressure = true; + } + + foreach (List pool in _pools.Values) + { + for (int i = pool.Count - 1; i >= 0; i--) + { + MemoryBlock block = pool[i]; + if (block.EmptySinceFrame < 0 || !block.IsEmpty) continue; + if (!pressure && _frame - block.EmptySinceFrame < EmptyBlockFrames) continue; + + pool.RemoveAt(i); + _emptyBlocks--; + _emptyBlocksFreed++; + NoteBlockReleased(block); + block.Dispose(); + } + } + } + } + + private ulong HeapBudgetLocked(int heap) => HeapBudgetOverrideForTests ?? _heapBudget[heap]; + + private ulong ReBarCapLocked(uint typeIndex) + { + if (ReBarCapOverrideForTests is { } forced) return forced; + uint heap = _memoryProperties.MemoryTypes[(int)typeIndex].HeapIndex; + return Math.Min(ReBarCapCeiling, HeapBudgetLocked((int)heap) / 4); + } + + private void RefreshBudgetLocked() + { + int heaps = (int)_memoryProperties.MemoryHeapCount; + if (BudgetExtension) + { + var budget = new PhysicalDeviceMemoryBudgetPropertiesEXT + { + SType = StructureType.PhysicalDeviceMemoryBudgetPropertiesExt, + }; + var properties = new PhysicalDeviceMemoryProperties2 + { + SType = StructureType.PhysicalDeviceMemoryProperties2, + PNext = &budget, + }; + _context.Api.GetPhysicalDeviceMemoryProperties2(_context.PhysicalDevice, &properties); + for (int i = 0; i < heaps; i++) + { + ulong reported = budget.HeapBudget[i]; + _heapBudget[i] = reported > 0 + ? reported + : (ulong)(_memoryProperties.MemoryHeaps[i].Size * FallbackBudgetShare); + } + return; + } + + for (int i = 0; i < heaps; i++) + { + _heapBudget[i] = (ulong)(_memoryProperties.MemoryHeaps[i].Size * FallbackBudgetShare); + } + } + + public MemorySnapshot Snapshot() + { + lock (_gate) + { + var heapBudget = new ulong[_heapBudget.Length]; + for (int i = 0; i < heapBudget.Length; i++) heapBudget[i] = HeapBudgetLocked(i); + + ulong cap = 0; + if (TryFindMemoryType(uint.MaxValue, + MemoryPropertyFlags.DeviceLocalBit | MemoryPropertyFlags.HostVisibleBit, 0, out uint reBarType)) + { + cap = ReBarCapLocked(reBarType); + } + + return new MemorySnapshot( + BlockCountLocked(), _dedicated.Count, _reBarUsed, cap, _reBarMisses, _emptyBlocksFreed, + BudgetExtension, (ulong[])_classBytes.Clone(), (ulong[])_heapUsed.Clone(), heapBudget); + } + } + + /// + /// Flags a request should avoid when it can: device-local-only memory keeps off + /// host-visible types (so ReBAR is left for per-frame data), and host memory + /// keeps off device-local types (so it does not eat the BAR either). + /// + private static MemoryPropertyFlags Avoided(MemoryPropertyFlags properties) + { + bool deviceLocal = (properties & MemoryPropertyFlags.DeviceLocalBit) != 0; + bool hostVisible = (properties & MemoryPropertyFlags.HostVisibleBit) != 0; + if (deviceLocal && !hostVisible) return MemoryPropertyFlags.HostVisibleBit; + if (hostVisible && !deviceLocal) return MemoryPropertyFlags.DeviceLocalBit; + return 0; + } + + private bool TryFindMemoryType(uint typeBits, MemoryPropertyFlags properties, MemoryPropertyFlags avoid, + out uint typeIndex) + { + for (uint i = 0; i < _memoryProperties.MemoryTypeCount; i++) + { + if ((typeBits & (1u << (int)i)) == 0) continue; + + MemoryPropertyFlags flags = _memoryProperties.MemoryTypes[(int)i].PropertyFlags; + if ((flags & properties) == properties && (flags & avoid) == 0) + { + typeIndex = i; + return true; + } + } + + typeIndex = 0; + return false; + } + + /// + /// The first type with every requested property, preferring one without the + /// avoided flags; on a unified-memory device nothing can be avoided and the + /// first match is taken. + /// + private uint FindMemoryType(uint typeBits, MemoryPropertyFlags properties, MemoryPropertyFlags avoid) + { + if (avoid != 0 && TryFindMemoryType(typeBits, properties, avoid, out uint preferred)) return preferred; + if (TryFindMemoryType(typeBits, properties, 0, out uint any)) return any; + throw new InvalidOperationException($"no memory type with {properties}"); + } + + /// The property flags of a memory type. Tests and diagnostics. + public MemoryPropertyFlags FlagsOf(uint typeIndex) => _memoryProperties.MemoryTypes[(int)typeIndex].PropertyFlags; + + /// Whether any type the mask allows has the properties and none of the avoided flags. + public bool HasMemoryType(uint typeBits, MemoryPropertyFlags properties, MemoryPropertyFlags avoid) + { + lock (_gate) return TryFindMemoryType(typeBits, properties, avoid, out _); + } + + /// A buffer's requirements plus whether the driver requires or prefers a dedicated allocation. + public static MemoryRequirements BufferRequirements(VulkanContext context, Buffer buffer, out bool dedicated) + { + var dedicatedRequirements = new MemoryDedicatedRequirements + { + SType = StructureType.MemoryDedicatedRequirements, + }; + var requirements = new MemoryRequirements2 + { + SType = StructureType.MemoryRequirements2, + PNext = &dedicatedRequirements, + }; + var info = new BufferMemoryRequirementsInfo2 + { + SType = StructureType.BufferMemoryRequirementsInfo2, + Buffer = buffer, + }; + context.Api.GetBufferMemoryRequirements2(context.Device, &info, &requirements); + dedicated = dedicatedRequirements.RequiresDedicatedAllocation || dedicatedRequirements.PrefersDedicatedAllocation; + return requirements.MemoryRequirements; + } + + /// An image's requirements plus whether the driver requires or prefers a dedicated allocation. + public static MemoryRequirements ImageRequirements(VulkanContext context, Image image, out bool dedicated) + { + var dedicatedRequirements = new MemoryDedicatedRequirements + { + SType = StructureType.MemoryDedicatedRequirements, + }; + var requirements = new MemoryRequirements2 + { + SType = StructureType.MemoryRequirements2, + PNext = &dedicatedRequirements, + }; + var info = new ImageMemoryRequirementsInfo2 + { + SType = StructureType.ImageMemoryRequirementsInfo2, + Image = image, + }; + context.Api.GetImageMemoryRequirements2(context.Device, &info, &requirements); + dedicated = dedicatedRequirements.RequiresDedicatedAllocation || dedicatedRequirements.PrefersDedicatedAllocation; + return requirements.MemoryRequirements; + } + + /// The stats.memory line: blocks, ReBAR use and misses, bytes per class, used/budget per heap. + public static string FormatMemoryLine(MemorySnapshot snapshot) + { + var line = new StringBuilder("stats.memory"); + line.Append(" blocks=").Append(snapshot.Blocks.ToString(CultureInfo.InvariantCulture)); + line.Append(" dedicated=").Append(snapshot.DedicatedBlocks.ToString(CultureInfo.InvariantCulture)); + line.Append(" rebar_used=").Append(snapshot.ReBarUsed.ToString(CultureInfo.InvariantCulture)); + line.Append(" rebar_cap=").Append(snapshot.ReBarCap.ToString(CultureInfo.InvariantCulture)); + line.Append(" rebar_misses=").Append(snapshot.ReBarMisses.ToString(CultureInfo.InvariantCulture)); + line.Append(" empty_blocks_freed=").Append(snapshot.EmptyBlocksFreed.ToString(CultureInfo.InvariantCulture)); + line.Append(" budget_ext=").Append(snapshot.BudgetExtension ? '1' : '0'); + line.Append(" class_bytes="); + for (int i = 0; i < PoolClassCount; i++) + { + if (i > 0) line.Append(','); + ulong bytes = snapshot.ClassBytes != null && i < snapshot.ClassBytes.Length ? snapshot.ClassBytes[i] : 0; + line.Append(bytes.ToString(CultureInfo.InvariantCulture)); + } + line.Append(" heaps="); + int heaps = snapshot.HeapUsed?.Length ?? 0; + for (int i = 0; i < heaps; i++) + { + if (i > 0) line.Append(','); + line.Append(snapshot.HeapUsed![i].ToString(CultureInfo.InvariantCulture)).Append('/'); + ulong budget = snapshot.HeapBudget != null && i < snapshot.HeapBudget.Length ? snapshot.HeapBudget[i] : 0; + line.Append(budget.ToString(CultureInfo.InvariantCulture)); + } + return line.ToString(); + } + + public void Dispose() + { + lock (_gate) + { + if (_disposed) return; + _disposed = true; + + foreach (List pool in _pools.Values) + { + foreach (MemoryBlock block in pool) block.Dispose(); + } + _pools.Clear(); + + foreach (MemoryBlock block in _dedicated) block.Dispose(); + _dedicated.Clear(); + } + } +} diff --git a/Optimum.Render.Vulkan/Core/VulkanCapabilities.cs b/Optimum.Render.Vulkan/Core/VulkanCapabilities.cs new file mode 100644 index 00000000..07b52989 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VulkanCapabilities.cs @@ -0,0 +1,162 @@ +using System; +using System.Collections.Generic; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// How a draw's effective colour write mask reaches the GPU (Phase 2, contract C4). +/// +/// The effective mask of attachment i is drawBufferEnabled(i) ? colorMask : 0, +/// then masked by the outputs the program writes. glDrawBuffers and the TAA motion +/// windows change it between draws of one pass; every tier expresses that without +/// restarting the rendering scope. Ordered from most to least capable. +/// +internal enum ColorWriteTier +{ + /// + /// The write-mask set is interned into the pipeline key: a mask change selects + /// another pipeline. Always available; bounded by programs used inside a window x 2. + /// + PipelineKey = 0, + + /// + /// VK_EXT_extended_dynamic_state3 colorWriteMask: the full per-attachment mask is + /// dynamic, so neither glColorMask nor glDrawBuffers is in the key. With + /// colorBlendEnable + colorBlendEquation also present the blend set leaves the + /// key as well. + /// + DynamicMask = 1, + + /// + /// VK_EXT_color_write_enable: glDrawBuffers is exactly a per-attachment enable, + /// zero extra pipelines; glColorMask stays in the key as before. + /// + DynamicEnable = 2, +} + +/// The optional-feature tier table: selection with an env override that forces a fallback. +internal static class DeviceCaps +{ + /// enable | mask | pipeline. A forced tier the device lacks degrades to the next one below. + public const string ColorWriteTierVariable = "OPTIMUM_VULKAN_COLOR_WRITE_TIER"; + + /// Parses an override; null for empty or unknown values. + public static ColorWriteTier? ParseColorWriteTier(string? value) => + value?.Trim().ToLowerInvariant() switch + { + "enable" or "dynamic-enable" => ColorWriteTier.DynamicEnable, + "mask" or "dynamic-mask" => ColorWriteTier.DynamicMask, + "pipeline" or "pipeline-key" or "key" => ColorWriteTier.PipelineKey, + _ => null, + }; + + /// + /// The best tier the device supports, or the forced one when it is supported; + /// a forced tier the device lacks falls to the best supported tier below it. + /// + public static ColorWriteTier SelectColorWriteTier(bool colorWriteEnable, bool colorWriteMask, ColorWriteTier? forced) + { + ColorWriteTier ceiling = forced ?? ColorWriteTier.DynamicEnable; + if (ceiling == ColorWriteTier.DynamicEnable && colorWriteEnable) return ColorWriteTier.DynamicEnable; + if (ceiling >= ColorWriteTier.DynamicMask && colorWriteMask) return ColorWriteTier.DynamicMask; + return ColorWriteTier.PipelineKey; + } + + /// The token written to the device log line and the stats. + public static string Token(ColorWriteTier tier) => tier switch + { + ColorWriteTier.DynamicEnable => "enable", + ColorWriteTier.DynamicMask => "mask", + _ => "pipeline", + }; + + public static ColorWriteTier? FromEnvironment() => + ParseColorWriteTier(Environment.GetEnvironmentVariable(ColorWriteTierVariable)); +} + +/// +/// What a device reports for the bindless texture model of plan decision 9, read +/// from VkPhysicalDeviceVulkan12Features/Properties and the 1.0 features and limits. +/// +internal readonly record struct DescriptorIndexingSupport( + bool RuntimeDescriptorArray, + bool DescriptorBindingPartiallyBound, + bool DescriptorBindingSampledImageUpdateAfterBind, + bool ShaderSampledImageArrayDynamicIndexing, + uint MaxPerStageDescriptorUpdateAfterBindSampledImages, + uint MaxPerStageDescriptorUpdateAfterBindSamplers, + uint MaxDescriptorSetUpdateAfterBindSampledImages, + uint MaxDescriptorSetUpdateAfterBindSamplers, + uint MaxDescriptorSetUpdateAfterBindUniformBuffersDynamic, + uint MaxPushConstantsSize); + +/// +/// The device floor for decision 9: one pipeline layout whose set 1 holds +/// combined-image-sampler arrays with PARTIALLY_BOUND | UPDATE_AFTER_BIND, indexed +/// per draw from push constants (docs/vulkan.md#bindless-descriptors, "Limits check at +/// startup" and "Features to enable"). A device below it is not used for Vulkan at +/// all - the session stays on OpenGL - rather than running a second, per-program +/// layout path. +/// +internal static class DescriptorIndexingFloor +{ + /// + /// Sum of the set-1 array capacities (, the + /// starting sizes in docs/vulkan.md#bindless-descriptors "Implementation for this + /// renderer"). + /// + public const uint BindlessSampledImages = Shaders.SetConvention.TextureArrayCapacityTotal; + + /// Set 0's fixed frame textures, with headroom (the plan names five). + public const uint FrameTextures = 16; + + /// + /// Combined image samplers count against both the sampled-image and the sampler + /// limits, and the per-stage update-after-bind limits count every set in the + /// layout, set 0 included. + /// + public const uint RequiredSampledImages = BindlessSampledImages + FrameTextures; + + /// + /// Set 0's frame block and set 2's program record are both dynamic uniform buffers in + /// the layout that also holds the update-after-bind set. + /// + public const uint RequiredDynamicUniformBuffers = 2; + + /// The spec minimum, and the budget decision 9 gives per-draw indices and scalars. + public const uint RequiredPushConstantBytes = 128; + + /// + /// Every requirement the device misses, as the feature or limit name the spec uses + /// (limits with the reported and the required value). Empty when the floor is met. + /// + public static List Missing(in DescriptorIndexingSupport support) + { + var missing = new List(); + if (!support.RuntimeDescriptorArray) missing.Add("runtimeDescriptorArray"); + if (!support.DescriptorBindingPartiallyBound) missing.Add("descriptorBindingPartiallyBound"); + if (!support.DescriptorBindingSampledImageUpdateAfterBind) missing.Add("descriptorBindingSampledImageUpdateAfterBind"); + // A per-draw index from a push constant is dynamic but uniform over the draw. + if (!support.ShaderSampledImageArrayDynamicIndexing) missing.Add("shaderSampledImageArrayDynamicIndexing"); + + AtLeast(missing, "maxPerStageDescriptorUpdateAfterBindSampledImages", + support.MaxPerStageDescriptorUpdateAfterBindSampledImages, RequiredSampledImages); + AtLeast(missing, "maxPerStageDescriptorUpdateAfterBindSamplers", + support.MaxPerStageDescriptorUpdateAfterBindSamplers, RequiredSampledImages); + AtLeast(missing, "maxDescriptorSetUpdateAfterBindSampledImages", + support.MaxDescriptorSetUpdateAfterBindSampledImages, RequiredSampledImages); + AtLeast(missing, "maxDescriptorSetUpdateAfterBindSamplers", + support.MaxDescriptorSetUpdateAfterBindSamplers, RequiredSampledImages); + // Set 0's frame UBO and set 2's program record are dynamic and share the layout + // with the update-after-bind set. + AtLeast(missing, "maxDescriptorSetUpdateAfterBindUniformBuffersDynamic", + support.MaxDescriptorSetUpdateAfterBindUniformBuffersDynamic, RequiredDynamicUniformBuffers); + AtLeast(missing, "maxPushConstantsSize", support.MaxPushConstantsSize, RequiredPushConstantBytes); + return missing; + } + + private static void AtLeast(List missing, string name, uint reported, uint required) + { + if (reported < required) missing.Add($"{name} {reported} < {required}"); + } +} diff --git a/Optimum.Render.Vulkan/Core/VulkanContext.cs b/Optimum.Render.Vulkan/Core/VulkanContext.cs new file mode 100644 index 00000000..edc56e3f --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VulkanContext.cs @@ -0,0 +1,1366 @@ +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using Silk.NET.Core; +using Silk.NET.Core.Native; +using Silk.NET.Vulkan; +using Silk.NET.Vulkan.Extensions.EXT; + +namespace Optimum.Render.Vulkan.Core; + +/// How the context should be brought up. +internal sealed class VulkanContextOptions +{ + /// No surface, no swapchain. Used by tests and capability probes. + public bool Headless; + + /// Turns on the validation layers and the debug messenger. + public bool EnableValidation; + + /// Comma list of extra layer checks: sync, best, mobile, gpu, gpu-only (). + public string ValidationFeatures = ""; + + /// Pins a physical device by index; -1 picks automatically. + public int PreferredDeviceIndex = -1; + + /// Instance extensions the window system needs (from GLFW). + public string[] RequiredInstanceExtensions = Array.Empty(); + + /// Called with each validation message when validation is on. + public Action? DebugCallback; + + /// + /// Fills freshly created images and host-visible buffers with a loud value + /// before first use (see ). Null reads + /// OPTIMUM_VULKAN_POISON once, at context creation. + /// + public bool? Poison; + + /// + /// Forces a colour write tier (); null reads + /// OPTIMUM_VULKAN_COLOR_WRITE_TIER. A tier the device lacks degrades to the next below. + /// + public ColorWriteTier? ColorWriteTier; + + /// + /// Tests only: sleeps this long before every vkAcquireNextImageKHR, standing + /// in for a compositor that holds images back (PresentDecouplingTests). + /// + public TimeSpan AcquireDelayForTests; +} + +/// What the chosen device can do, once it is up. +internal sealed class VulkanCapabilities +{ + public string DeviceName = ""; + public string DriverName = ""; + public uint VendorId; + public uint DeviceId; + public uint DriverVersion; + /// VkPhysicalDeviceProperties::pipelineCacheUUID; 16 bytes. + public byte[] PipelineCacheUuid = new byte[16]; + public uint ApiVersion; + public PhysicalDeviceType DeviceType; + public uint MaxImageDimension2D; + public bool WideLines; + + /// + /// VkPhysicalDeviceLimits::lineWidthRange, the only widths vkCmdSetLineWidth accepts + /// once wideLines is on. GL silently clamps glLineWidth to its own range; Vulkan makes an + /// out-of-range width a validation error, and the game asks for 0.5 (the aiming reticle's + /// accuracy rectangle) and 2 (the camera path), so both routes clamp through + /// rather than passing the caller's value on. + /// + public float LineWidthMin = 1.0f; + public float LineWidthMax = 1.0f; + + /// + /// The width a line draw may actually rasterize with: the caller's, clamped to the device's + /// range, or exactly 1 on a device without wideLines. Applied to a native pipeline's dynamic + /// state, so every draw clamps the same way. + /// + public float ClampLineWidth(float width) + { + if (!WideLines) return 1.0f; + if (width < LineWidthMin) return LineWidthMin; + if (width > LineWidthMax) return LineWidthMax; + return width; + } + + public bool FillModeNonSolid; + public bool SamplerAnisotropy; + public bool MultiDrawIndirect; + /// Enabled whenever available; occlusion queries then count samples exactly, like GL_SAMPLES_PASSED. + public bool OcclusionQueryPrecise; + public float MaxSamplerLodBias; + public int MaxBoundDescriptorSets; + public ulong MinUniformBufferOffsetAlignment; + public ulong MinStorageBufferOffsetAlignment; + public ulong MaxUniformBufferRange; + public uint MaxColorAttachments = 8; + /// The bindless features and limits of plan decision 9; every selected device meets . + public DescriptorIndexingSupport DescriptorIndexing; + + /// VK_EXT_color_write_enable enabled (only when the selected tier uses it). + public bool ColorWriteEnable; + /// VK_EXT_extended_dynamic_state3 colorWriteMask enabled (only for the mask tier). + public bool DynamicColorWriteMask; + /// colorBlendEnable + colorBlendEquation enabled alongside the mask tier: the blend set is dynamic. + public bool DynamicColorBlend; + /// The tier draws use; see . + public ColorWriteTier ColorWriteTier = ColorWriteTier.PipelineKey; + + /// VK_KHR_present_id is enabled with its feature, so Swapchain.Present may chain VkPresentIdKHR. + public bool PresentIdEnabled; + + /// Swapchain maintenance is enabled, allowing explicit presentation fences. + public bool PresentFencesEnabled; + + /// The latency part of the "device up" log line. + public string LatencySummary => "CPU frame timing, present id " + (PresentIdEnabled ? "ON" : "OFF") + + ", present fences " + (PresentFencesEnabled ? "ON" : "OFF"); + /// + /// pipelineCreationCacheControl (core in 1.3, optional to support) enabled: pipelines can be + /// created with FAIL_ON_PIPELINE_COMPILE_REQUIRED, which the background compile path needs. + /// + public bool PipelineCreationCacheControl; +} + +/// +/// Instance, physical device, logical device and queue. +/// +/// Device selection and feature negotiation are the one place this backend is +/// allowed to give up. Anything missing here means the session falls back to +/// OpenGL with a logged reason rather than failing, so every check reports +/// instead of throwing. +/// +internal sealed unsafe class VulkanContext : IDisposable +{ + /// + /// The floor. 1.3 makes dynamic rendering and synchronization2 core, which + /// removes render-pass and framebuffer objects from the design entirely, and + /// brings the dynamic pipeline state that keeps the pipeline cache small. + /// Everything below it stays on OpenGL. + /// + public static readonly uint MinimumApiVersion = Vk.Version13; + + public Vk Api { get; private set; } = null!; + public Instance Instance { get; private set; } + public PhysicalDevice PhysicalDevice { get; private set; } + public Device Device { get; private set; } + public Queue GraphicsQueue { get; private set; } + public uint GraphicsQueueFamily { get; private set; } + + /// + /// Guards every submission to . + /// + /// Vulkan requires a queue to be externally synchronised: vkQueueSubmit and + /// vkQueuePresentKHR from two threads at once is undefined behaviour, and in + /// practice loses the device. The client does exactly that - asset loading + /// uploads textures off the main thread, each upload being its own + /// submit-and-wait, while the render thread is submitting frames. GL made + /// this impossible by having one context on one thread; here it has to be + /// enforced. + /// + public object QueueLock { get; } = new(); + + /// + /// Backs every buffer and image out of a few large blocks. See + /// for why one allocation per resource is not + /// an option. + /// + public VulkanAllocator Allocator { get; private set; } = null!; + + /// + /// VK_EXT_memory_budget is enabled, so the allocator reads per-heap budgets + /// from the driver. Off when the device lacks it or OPTIMUM_VULKAN_NO_MEMORY_BUDGET=1 + /// forces the heap x 0.7 fallback. + /// + public bool MemoryBudgetAvailable { get; private set; } + public VulkanCapabilities Capabilities { get; private set; } = new(); + + /// vkCmdSetColorWriteEnableEXT, when the enable tier is selected. + public ExtColorWriteEnable? ColorWriteEnableApi { get; private set; } + + /// vkCmdSetColorWriteMaskEXT / BlendEnable / BlendEquation, when the mask tier is selected. + public ExtExtendedDynamicState3? DynamicState3Api { get; private set; } + + /// + /// Whether the validation layers are actually loaded, which is not the same + /// as having been asked for: the layer has to be installed on the machine. + /// Worth being able to check, because "no validation messages" otherwise + /// reads as "nothing is wrong". + /// + public bool ValidationEnabled { get; private set; } + + /// + /// Whether freshly created images and host-visible buffers are filled with + /// values. Fixed for the context's life. + /// + public bool PoisonFreshResources { get; private set; } + + /// Tests only; see . + public TimeSpan AcquireDelayForTests { get; private set; } + + /// OPTIMUM_VULKAN_POISON: any value but empty and "0" turns poison mode on. + public const string PoisonVariable = "OPTIMUM_VULKAN_POISON"; + + internal static bool PoisonRequested(string? setting) => + !string.IsNullOrWhiteSpace(setting) && setting.Trim() != "0"; + + /// Marks a diagnostic the layers reported at error severity. + public const string ErrorPrefix = "[error] "; + + /// + /// Whether the driver records GPU checkpoints (VK_NV_device_diagnostic_checkpoints). + /// + /// A device loss otherwise says only that the GPU gave up. With checkpoints + /// the driver also reports the last marker each pipeline stage reached, which + /// names the draw or copy it was executing when it stopped. NVIDIA only; on + /// by default where present, off with OPTIMUM_VULKAN_CHECKPOINTS=0. + /// + public bool CheckpointsAvailable { get; private set; } + + /// Whether VK_EXT_device_fault can describe a loss after the fact. + public bool DeviceFaultAvailable { get; private set; } + + // Loaded by address rather than through an extension package: two entry + // points do not justify a dependency and another native DLL to ship. + private nint _cmdSetCheckpoint; + private nint _getQueueCheckpointData; + private ExtDeviceFault? _deviceFault; + + /// The instance extensions actually named in VkInstanceCreateInfo. + public string[] EnabledInstanceExtensions { get; private set; } = Array.Empty(); + + /// The device extensions actually named in VkDeviceCreateInfo. + public string[] EnabledDeviceExtensions { get; private set; } = Array.Empty(); + + private ExtDebugUtils? _debugUtils; + private DebugUtilsMessengerEXT _debugMessenger; + private Action? _debugCallback; + private PfnDebugUtilsMessengerCallbackEXT _debugDelegate; + private bool _disposed; + + private const string ValidationLayer = "VK_LAYER_KHRONOS_validation"; + + /// + /// Brings the context up, or explains why it cannot. Never throws for an + /// ordinary unsupported-hardware outcome. + /// + public static bool TryCreate( + VulkanContextOptions options, out VulkanContext? context, out string? failureReason) + { + context = null; + failureReason = null; + + var created = new VulkanContext(); + created.AcquireDelayForTests = options.AcquireDelayForTests; + created.PoisonFreshResources = options.Poison + ?? PoisonRequested(Environment.GetEnvironmentVariable(PoisonVariable)); + try + { + created.Api = Vk.GetApi(); + } + catch (Exception error) + { + failureReason = "no Vulkan loader: " + error.Message; + return false; + } + + try + { + if (!created.CreateInstance(options, out failureReason)) { created.Dispose(); return false; } + if (!created.SelectPhysicalDevice(options, out failureReason)) { created.Dispose(); return false; } + if (!created.CreateDevice(options, out failureReason)) { created.Dispose(); return false; } + } + catch (Exception error) + { + failureReason = error.Message; + created.Dispose(); + return false; + } + + context = created; + return true; + } + + // ------------------------------------------------------------------ instance + + private bool CreateInstance(VulkanContextOptions options, out string? failureReason) + { + failureReason = null; + + uint loaderVersion = Vk.Version10; + if (Api.EnumerateInstanceVersion(ref loaderVersion) != Result.Success) + { + loaderVersion = Vk.Version10; + } + if (loaderVersion < MinimumApiVersion) + { + failureReason = + $"Vulkan loader reports {VersionString(loaderVersion)}, " + + $"but {VersionString(MinimumApiVersion)} is required"; + return false; + } + + var extensions = new List(options.RequiredInstanceExtensions); + // Surface maintenance dependencies are instance extensions; enable them before + // choosing a device, then query that device's optional maintenance feature. + if (!options.Headless && extensions.Contains("VK_KHR_surface") + && LayerAdvertisesExtension(Api, null, "VK_KHR_get_surface_capabilities2")) + { + foreach (string maintenance in new[] { "VK_KHR_surface_maintenance1", "VK_EXT_surface_maintenance1" }) + { + if (!LayerAdvertisesExtension(Api, null, maintenance)) continue; + if (!extensions.Contains("VK_KHR_get_surface_capabilities2")) extensions.Add("VK_KHR_get_surface_capabilities2"); + if (!extensions.Contains(maintenance)) extensions.Add(maintenance); + } + } + bool validation = options.EnableValidation && HasValidationLayer(); + + // Extra layer checks (sync validation, best practices with the vendor sets, + // GPU-assisted validation) go through VK_EXT_layer_settings, which replaces + // the deprecated VK_EXT_validation_features (docs/vulkan.md#validation-and-acceptance + // §1); a layer too old for it still gets the deprecated struct. Both are + // decided before the extension list is marshalled, because an extension has + // to be enabled under exactly the condition its struct is chained - a chained + // struct whose extension was never enabled is ignored at best. And only when + // the layer advertises it: naming an extension the layer does not have fails + // vkCreateInstance outright, which would turn a diagnostic environment + // variable into a silent fall back to OpenGL (rule 1). + List settings = ValidationLayerSettings(options.ValidationFeatures); + List enables = ParseValidationFeatures(options.ValidationFeatures); + bool chainLayerSettings = validation && settings.Count > 0 + && LayerAdvertisesExtension(Api, ValidationLayer, LayerSettingsExtensionName); + bool chainValidationFeatures = validation && !chainLayerSettings && enables.Count > 0 + && LayerAdvertisesExtension(Api, ValidationLayer, ValidationFeaturesExtensionName); + if (validation) + { + extensions.Add(ExtDebugUtils.ExtensionName); + } + if (chainLayerSettings) + { + extensions.Add(LayerSettingsExtensionName); + } + if (chainValidationFeatures) + { + extensions.Add(ValidationFeaturesExtensionName); + } + ValidationSettingsApplied = + chainLayerSettings ? "layer settings " + DescribeValidationLayerSettings(settings) + : chainValidationFeatures ? "deprecated validation features " + string.Join(",", enables) + : validation && settings.Count > 0 + ? "extra checks NOT APPLIED (the layer has neither VK_EXT_layer_settings nor VK_EXT_validation_features)" + : ""; + + EnabledInstanceExtensions = extensions.ToArray(); + + byte* applicationName = (byte*)SilkMarshal.StringToPtr("Optimum"); + byte* engineName = (byte*)SilkMarshal.StringToPtr("Optimum.Render.Vulkan"); + nint extensionsPtr = SilkMarshal.StringArrayToPtr(extensions); + nint layersPtr = validation ? SilkMarshal.StringArrayToPtr(new[] { ValidationLayer }) : 0; + // Every string a layer setting points at, freed with the rest below. + var settingStrings = new List(); + + try + { + int settingSlots = Math.Max(settings.Count, 1); + LayerSettingEXT* settingsPtr = stackalloc LayerSettingEXT[settingSlots]; + Bool32* boolValues = stackalloc Bool32[settingSlots]; + nint* textValues = stackalloc nint[settingSlots]; + if (chainLayerSettings) + { + nint layerName = SilkMarshal.StringToPtr(ValidationLayer); + settingStrings.Add(layerName); + for (int i = 0; i < settings.Count; i++) + { + nint settingName = SilkMarshal.StringToPtr(settings[i].Name); + settingStrings.Add(settingName); + settingsPtr[i] = new LayerSettingEXT + { + PLayerName = (byte*)layerName, + PSettingName = (byte*)settingName, + ValueCount = 1, + }; + if (settings[i].Text == null) + { + boolValues[i] = settings[i].Enabled; + settingsPtr[i].Type = LayerSettingTypeEXT.Bool32Ext; + settingsPtr[i].PValues = &boolValues[i]; + } + else + { + textValues[i] = SilkMarshal.StringToPtr(settings[i].Text); + settingStrings.Add(textValues[i]); + settingsPtr[i].Type = LayerSettingTypeEXT.StringExt; + settingsPtr[i].PValues = &textValues[i]; + } + } + } + var layerSettings = new LayerSettingsCreateInfoEXT + { + SType = StructureType.LayerSettingsCreateInfoExt, + SettingCount = (uint)settings.Count, + PSettings = settingsPtr, + }; + + var applicationInfo = new ApplicationInfo + { + SType = StructureType.ApplicationInfo, + PApplicationName = applicationName, + ApplicationVersion = new Version32(1, 0, 0), + PEngineName = engineName, + EngineVersion = new Version32(1, 0, 0), + ApiVersion = MinimumApiVersion, + }; + + ValidationFeatureEnableEXT* enablesPtr = stackalloc ValidationFeatureEnableEXT[Math.Max(enables.Count, 1)]; + for (int i = 0; i < enables.Count; i++) enablesPtr[i] = enables[i]; + var validationFeatures = new ValidationFeaturesEXT + { + SType = StructureType.ValidationFeaturesExt, + EnabledValidationFeatureCount = (uint)enables.Count, + PEnabledValidationFeatures = enablesPtr, + }; + + var createInfo = new InstanceCreateInfo + { + SType = StructureType.InstanceCreateInfo, + PNext = chainLayerSettings ? &layerSettings + : chainValidationFeatures ? (void*)&validationFeatures + : null, + PApplicationInfo = &applicationInfo, + EnabledExtensionCount = (uint)extensions.Count, + PpEnabledExtensionNames = (byte**)extensionsPtr, + EnabledLayerCount = validation ? 1u : 0u, + PpEnabledLayerNames = validation ? (byte**)layersPtr : null, + }; + + Result result = Api.CreateInstance(&createInfo, null, out Instance instance); + if (result != Result.Success) + { + failureReason = "vkCreateInstance failed: " + result; + return false; + } + Instance = instance; + } + finally + { + SilkMarshal.Free((nint)applicationName); + SilkMarshal.Free((nint)engineName); + SilkMarshal.Free(extensionsPtr); + if (layersPtr != 0) SilkMarshal.Free(layersPtr); + foreach (nint text in settingStrings) SilkMarshal.Free(text); + } + + if (validation) + { + SetUpDebugMessenger(options); + } + + return true; + } + + /// Name of VK_EXT_validation_features; Silk.NET has no wrapper class for it. + internal const string ValidationFeaturesExtensionName = "VK_EXT_validation_features"; + + /// Name of VK_EXT_layer_settings, the layer's own replacement for it. + internal const string LayerSettingsExtensionName = "VK_EXT_layer_settings"; + + /// The validation layer's spec and implementation version, once it has been found. + public string ValidationLayerVersion { get; private set; } = ""; + + /// + /// Which extra checks reached the layer and how, for the device-up log line; empty when + /// none were asked for. + /// + public string ValidationSettingsApplied { get; private set; } = ""; + + /// One VK_EXT_layer_settings entry for the Khronos layer: a boolean, or a string when is set. + internal readonly record struct ValidationLayerSetting(string Name, bool Enabled = true, string? Text = null); + + /// + /// Maps the comma list from OPTIMUM_VULKAN_VALIDATION_FEATURES onto the layer's settings + /// (VkLayer_khronos_validation.json; docs/vulkan.md#validation-and-acceptance §1 and §2): + /// "sync" synchronization validation with structured message properties to filter on; + /// "best" best practices with the NVIDIA and AMD sets, whose messages are warnings and + /// performance reports; "mobile" the Arm and IMG sets, advisory on desktop GPUs; + /// "gpu" GPU-assisted validation; "gpu-only" the same with CPU core validation off. + /// + internal static List ValidationLayerSettings(string? features) + { + var settings = new List(); + foreach (string feature in (features ?? "").Split(',', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)) + { + switch (feature.ToLowerInvariant()) + { + case "sync": + AddSetting(settings, new ValidationLayerSetting("validate_sync")); + AddSetting(settings, new ValidationLayerSetting("syncval_message_extra_properties")); + break; + case "best": + AddSetting(settings, new ValidationLayerSetting("validate_best_practices")); + AddSetting(settings, new ValidationLayerSetting("validate_best_practices_nvidia")); + AddSetting(settings, new ValidationLayerSetting("validate_best_practices_amd")); + AddSetting(settings, new ValidationLayerSetting("report_flags", Text: "error,warn,perf")); + break; + case "mobile": + AddSetting(settings, new ValidationLayerSetting("validate_best_practices")); + AddSetting(settings, new ValidationLayerSetting("validate_best_practices_arm")); + AddSetting(settings, new ValidationLayerSetting("validate_best_practices_img")); + break; + case "gpu": + AddSetting(settings, new ValidationLayerSetting("gpuav_enable")); + break; + case "gpu-only": + AddSetting(settings, new ValidationLayerSetting("gpuav_enable")); + AddSetting(settings, new ValidationLayerSetting("validate_core", Enabled: false)); + break; + } + } + return settings; + } + + private static void AddSetting(List settings, ValidationLayerSetting setting) + { + foreach (ValidationLayerSetting existing in settings) + { + if (existing.Name == setting.Name) return; + } + settings.Add(setting); + } + + internal static string DescribeValidationLayerSettings(List settings) + { + var parts = new List(settings.Count); + foreach (ValidationLayerSetting setting in settings) + { + parts.Add(setting.Text != null ? setting.Name + "=" + setting.Text + : setting.Enabled ? setting.Name + : setting.Name + "=false"); + } + return string.Join(" ", parts); + } + + /// + /// Whether advertises + /// as an instance extension. A layer's extensions are invisible to the + /// loader-level enumeration, so the layer has to be named explicitly. Null queries + /// the loader-level extensions instead. + /// + internal static bool LayerAdvertisesExtension(Vk api, string? layerName, string extensionName) + { + nint layer = layerName == null ? 0 : SilkMarshal.StringToPtr(layerName); + try + { + uint count = 0; + if (api.EnumerateInstanceExtensionProperties((byte*)layer, &count, null) != Result.Success + || count == 0) + { + return false; + } + + var properties = new ExtensionProperties[count]; + fixed (ExtensionProperties* propertiesPtr = properties) + { + if (api.EnumerateInstanceExtensionProperties((byte*)layer, &count, propertiesPtr) != Result.Success) + { + return false; + } + for (int i = 0; i < count; i++) + { + // The name is a fixed-size buffer, readable only through a pointer. + if (SilkMarshal.PtrToString((nint)propertiesPtr[i].ExtensionName) == extensionName) + { + return true; + } + } + } + return false; + } + finally + { + if (layer != 0) SilkMarshal.Free(layer); + } + } + + /// Maps the comma list from OPTIMUM_VULKAN_VALIDATION_FEATURES onto layer feature flags. + internal static List ParseValidationFeatures(string? features) + { + var enables = new List(); + foreach (string feature in (features ?? "").Split(',', StringSplitOptions.RemoveEmptyEntries | StringSplitOptions.TrimEntries)) + { + switch (feature.ToLowerInvariant()) + { + case "sync": enables.Add(ValidationFeatureEnableEXT.SynchronizationValidationExt); break; + case "best": enables.Add(ValidationFeatureEnableEXT.BestPracticesExt); break; + case "mobile": enables.Add(ValidationFeatureEnableEXT.BestPracticesExt); break; + case "gpu": enables.Add(ValidationFeatureEnableEXT.GpuAssistedExt); break; + case "gpu-only": enables.Add(ValidationFeatureEnableEXT.GpuAssistedExt); break; + } + } + return enables; + } + + private bool HasValidationLayer() + { + uint count = 0; + if (Api.EnumerateInstanceLayerProperties(ref count, null) != Result.Success || count == 0) + { + return false; + } + + var layers = new LayerProperties[count]; + fixed (LayerProperties* layersPtr = layers) + { + if (Api.EnumerateInstanceLayerProperties(ref count, layersPtr) != Result.Success) + { + return false; + } + + for (int i = 0; i < count; i++) + { + if (SilkMarshal.PtrToString((nint)layersPtr[i].LayerName) == ValidationLayer) + { + ValidationLayerVersion = VersionString(layersPtr[i].SpecVersion) + + " (implementation " + layersPtr[i].ImplementationVersion + ")"; + return true; + } + } + } + return false; + } + + private void SetUpDebugMessenger(VulkanContextOptions options) + { + if (!Api.TryGetInstanceExtension(Instance, out ExtDebugUtils debugUtils)) return; + + _debugUtils = debugUtils; + _debugCallback = options.DebugCallback; + ValidationEnabled = true; + _debugDelegate = new PfnDebugUtilsMessengerCallbackEXT(OnDebugMessage); + + var createInfo = new DebugUtilsMessengerCreateInfoEXT + { + SType = StructureType.DebugUtilsMessengerCreateInfoExt, + MessageSeverity = DebugUtilsMessageSeverityFlagsEXT.WarningBitExt + | DebugUtilsMessageSeverityFlagsEXT.ErrorBitExt, + MessageType = DebugUtilsMessageTypeFlagsEXT.GeneralBitExt + | DebugUtilsMessageTypeFlagsEXT.ValidationBitExt + | DebugUtilsMessageTypeFlagsEXT.PerformanceBitExt, + PfnUserCallback = _debugDelegate, + }; + + _debugUtils.CreateDebugUtilsMessenger(Instance, &createInfo, null, out _debugMessenger); + } + + private uint OnDebugMessage( + DebugUtilsMessageSeverityFlagsEXT severity, + DebugUtilsMessageTypeFlagsEXT types, + DebugUtilsMessengerCallbackDataEXT* data, + void* userData) + { + string? message = SilkMarshal.PtrToString((nint)data->PMessage); + if (message != null) + { + // Severity is prefixed rather than dropped. The layers report real + // spec violations alongside advisories - "this fragment output has + // no attachment and the write is unused" is a note, not a fault - + // and the client turns diagnostics into thrown exceptions through + // CheckGlError. Without the distinction every advisory would read as + // a GL error and abort a frame that was fine. + string prefix = severity.HasFlag(DebugUtilsMessageSeverityFlagsEXT.ErrorBitExt) + ? ErrorPrefix + : "[warning] "; + // The layer names the check separately (SYNC-HAZARD-WRITE-AFTER-WRITE, + // BestPractices-..., a VUID); current layers no longer repeat it in + // the text, and without it a log line cannot be grouped or pinned. + string? id = SilkMarshal.PtrToString((nint)data->PMessageIdName); + string idTag = string.IsNullOrEmpty(id) ? "" : "[" + id + "] "; + _debugCallback?.Invoke(prefix + idTag + message); + } + return Vk.False; + } + + // ----------------------------------------------------------- device selection + + private bool SelectPhysicalDevice(VulkanContextOptions options, out string? failureReason) + { + failureReason = null; + + uint count = 0; + Api.EnumeratePhysicalDevices(Instance, ref count, null); + if (count == 0) + { + failureReason = "no Vulkan physical devices"; + return false; + } + + var devices = new PhysicalDevice[count]; + fixed (PhysicalDevice* devicesPtr = devices) + { + Api.EnumeratePhysicalDevices(Instance, ref count, devicesPtr); + } + + if (options.PreferredDeviceIndex >= 0) + { + if (options.PreferredDeviceIndex >= devices.Length) + { + failureReason = $"device index {options.PreferredDeviceIndex} out of range ({devices.Length} present)"; + return false; + } + + PhysicalDevice pinned = devices[options.PreferredDeviceIndex]; + if (!IsUsable(pinned, out string? why)) + { + failureReason = $"pinned device is unusable: {why}"; + return false; + } + PhysicalDevice = pinned; + return true; + } + + // Prefer a discrete GPU, then integrated, then anything usable. The + // handheld this targets has only an integrated one; a desktop with both + // should get the fast one. + var rejections = new List(); + PhysicalDevice best = default; + int bestScore = -1; + + foreach (PhysicalDevice candidate in devices) + { + if (!IsUsable(candidate, out string? why)) + { + rejections.Add(why!); + continue; + } + + PhysicalDeviceProperties properties = Api.GetPhysicalDeviceProperties(candidate); + int score = properties.DeviceType switch + { + PhysicalDeviceType.DiscreteGpu => 3, + PhysicalDeviceType.IntegratedGpu => 2, + PhysicalDeviceType.VirtualGpu => 1, + _ => 0, + }; + + if (score > bestScore) + { + bestScore = score; + best = candidate; + } + } + + if (bestScore < 0) + { + failureReason = "no usable Vulkan device: " + string.Join("; ", rejections); + return false; + } + + PhysicalDevice = best; + return true; + } + + /// + /// A device is usable when it meets the API floor, exposes a graphics queue, + /// and supports the features the renderer is built on. + /// + private bool IsUsable(PhysicalDevice device, out string? reason) + { + PhysicalDeviceProperties properties = Api.GetPhysicalDeviceProperties(device); + string name = SilkMarshal.PtrToString((nint)properties.DeviceName) ?? "unknown"; + + if (properties.ApiVersion < MinimumApiVersion) + { + reason = $"{name} reports {VersionString(properties.ApiVersion)}, " + + $"below {VersionString(MinimumApiVersion)}"; + return false; + } + + if (!TryFindGraphicsQueue(device, out _)) + { + reason = $"{name} has no graphics queue family"; + return false; + } + + var vulkan13 = new PhysicalDeviceVulkan13Features { SType = StructureType.PhysicalDeviceVulkan13Features }; + var vulkan12 = new PhysicalDeviceVulkan12Features + { + SType = StructureType.PhysicalDeviceVulkan12Features, + PNext = &vulkan13, + }; + var features = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &vulkan12, + }; + Api.GetPhysicalDeviceFeatures2(device, &features); + + var missing = new List(); + if (!vulkan13.DynamicRendering) missing.Add("dynamicRendering"); + if (!vulkan13.Synchronization2) missing.Add("synchronization2"); + // Scalar layout is what lets the game's tightly packed float[] uniform + // uploads land in the generated block as a memcpy. + if (!vulkan12.ScalarBlockLayout) missing.Add("scalarBlockLayout"); + if (!vulkan12.TimelineSemaphore) missing.Add("timelineSemaphore"); + // The OIT and SSAO passes set blend state per attachment. + if (!features.Features.IndependentBlend) missing.Add("independentBlend"); + // Chunk rendering issues one indirect multidraw per pool. + if (!features.Features.MultiDrawIndirect) missing.Add("multiDrawIndirect"); + // Decision 9: one pipeline layout with bindless textures. A device without + // it stays on OpenGL rather than running a second, per-program layout path. + missing.AddRange(DescriptorIndexingFloor.Missing(ReadDescriptorIndexingSupport(device))); + + if (missing.Count > 0) + { + reason = $"{name} lacks {string.Join(", ", missing)}"; + return false; + } + + reason = null; + return true; + } + + /// The features and limits judges, as the device reports them. + private DescriptorIndexingSupport ReadDescriptorIndexingSupport(PhysicalDevice device) + { + var vulkan12Features = new PhysicalDeviceVulkan12Features { SType = StructureType.PhysicalDeviceVulkan12Features }; + var features = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &vulkan12Features, + }; + Api.GetPhysicalDeviceFeatures2(device, &features); + + var vulkan12Properties = new PhysicalDeviceVulkan12Properties { SType = StructureType.PhysicalDeviceVulkan12Properties }; + var properties = new PhysicalDeviceProperties2 + { + SType = StructureType.PhysicalDeviceProperties2, + PNext = &vulkan12Properties, + }; + Api.GetPhysicalDeviceProperties2(device, &properties); + + return new DescriptorIndexingSupport( + RuntimeDescriptorArray: vulkan12Features.RuntimeDescriptorArray, + DescriptorBindingPartiallyBound: vulkan12Features.DescriptorBindingPartiallyBound, + DescriptorBindingSampledImageUpdateAfterBind: vulkan12Features.DescriptorBindingSampledImageUpdateAfterBind, + ShaderSampledImageArrayDynamicIndexing: features.Features.ShaderSampledImageArrayDynamicIndexing, + MaxPerStageDescriptorUpdateAfterBindSampledImages: vulkan12Properties.MaxPerStageDescriptorUpdateAfterBindSampledImages, + MaxPerStageDescriptorUpdateAfterBindSamplers: vulkan12Properties.MaxPerStageDescriptorUpdateAfterBindSamplers, + MaxDescriptorSetUpdateAfterBindSampledImages: vulkan12Properties.MaxDescriptorSetUpdateAfterBindSampledImages, + MaxDescriptorSetUpdateAfterBindSamplers: vulkan12Properties.MaxDescriptorSetUpdateAfterBindSamplers, + MaxDescriptorSetUpdateAfterBindUniformBuffersDynamic: vulkan12Properties.MaxDescriptorSetUpdateAfterBindUniformBuffersDynamic, + MaxPushConstantsSize: properties.Properties.Limits.MaxPushConstantsSize); + } + + private bool TryFindGraphicsQueue(PhysicalDevice device, out uint family) + { + family = 0; + uint count = 0; + Api.GetPhysicalDeviceQueueFamilyProperties(device, ref count, null); + if (count == 0) return false; + + var families = new QueueFamilyProperties[count]; + fixed (QueueFamilyProperties* familiesPtr = families) + { + Api.GetPhysicalDeviceQueueFamilyProperties(device, ref count, familiesPtr); + } + + for (uint i = 0; i < count; i++) + { + if (families[i].QueueFlags.HasFlag(QueueFlags.GraphicsBit)) + { + family = i; + return true; + } + } + return false; + } + + // -------------------------------------------------------------- logical device + + private bool CreateDevice(VulkanContextOptions options, out string? failureReason) + { + failureReason = null; + + if (!TryFindGraphicsQueue(PhysicalDevice, out uint family)) + { + failureReason = "graphics queue family disappeared between selection and creation"; + return false; + } + GraphicsQueueFamily = family; + + float priority = 1.0f; + var queueCreateInfo = new DeviceQueueCreateInfo + { + SType = StructureType.DeviceQueueCreateInfo, + QueueFamilyIndex = family, + QueueCount = 1, + PQueuePriorities = &priority, + }; + + PhysicalDeviceFeatures available = Api.GetPhysicalDeviceFeatures(PhysicalDevice); + + // Diagnostics for a lost device. Both are optional and cost nothing when + // the GPU is healthy, so they are taken wherever the driver offers them. + Dictionary deviceExtensionsAvailable = EnumerateDeviceExtensions(); + bool checkpointsDisabled = + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_CHECKPOINTS") is "0" or "off" or "false"; + bool wantCheckpoints = !checkpointsDisabled && IntPtr.Size == 8 + && deviceExtensionsAvailable.ContainsKey("VK_NV_device_diagnostic_checkpoints"); + + var faultFeatures = new PhysicalDeviceFaultFeaturesEXT + { + SType = StructureType.PhysicalDeviceFaultFeaturesExt, + }; + bool wantDeviceFault = false; + if (deviceExtensionsAvailable.ContainsKey("VK_EXT_device_fault")) + { + var query = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &faultFeatures, + }; + Api.GetPhysicalDeviceFeatures2(PhysicalDevice, &query); + wantDeviceFault = faultFeatures.DeviceFault; + + // Re-request only the feature that is wanted; the query may have + // reported others this backend has no use for. + faultFeatures = new PhysicalDeviceFaultFeaturesEXT + { + SType = StructureType.PhysicalDeviceFaultFeaturesExt, + DeviceFault = wantDeviceFault, + }; + } + + var enabledFeatures = new PhysicalDeviceFeatures + { + IndependentBlend = true, + MultiDrawIndirect = true, + // Decision 9: per-draw bindless indices come from push constants. + ShaderSampledImageArrayDynamicIndexing = true, + // Optional. Wireframe debug and thick lines degrade rather than fail. + FillModeNonSolid = available.FillModeNonSolid, + WideLines = available.WideLines, + SamplerAnisotropy = available.SamplerAnisotropy, + DepthClamp = available.DepthClamp, + ShaderClipDistance = available.ShaderClipDistance, + OcclusionQueryPrecise = available.OcclusionQueryPrecise, + }; + + // Optional tier (Phase 2, C4): colour write masks as dynamic state. Only the + // extension the selected tier uses is enabled, so validation sees exactly + // what draws record. OPTIMUM_VULKAN_COLOR_WRITE_TIER forces a fallback. + var colorWriteFeatures = new PhysicalDeviceColorWriteEnableFeaturesEXT + { + SType = StructureType.PhysicalDeviceColorWriteEnableFeaturesExt, + }; + var dynamicState3Features = new PhysicalDeviceExtendedDynamicState3FeaturesEXT + { + SType = StructureType.PhysicalDeviceExtendedDynamicState3FeaturesExt, + }; + bool hasColorWriteEnable = deviceExtensionsAvailable.ContainsKey("VK_EXT_color_write_enable"); + bool hasDynamicState3 = deviceExtensionsAvailable.ContainsKey("VK_EXT_extended_dynamic_state3"); + if (hasColorWriteEnable || hasDynamicState3) + { + colorWriteFeatures.PNext = hasDynamicState3 ? &dynamicState3Features : null; + var query = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = hasColorWriteEnable ? &colorWriteFeatures : &dynamicState3Features, + }; + Api.GetPhysicalDeviceFeatures2(PhysicalDevice, &query); + } + bool canEnable = hasColorWriteEnable && colorWriteFeatures.ColorWriteEnable; + bool canMask = hasDynamicState3 && dynamicState3Features.ExtendedDynamicState3ColorWriteMask; + bool canBlend = canMask && dynamicState3Features.ExtendedDynamicState3ColorBlendEnable + && dynamicState3Features.ExtendedDynamicState3ColorBlendEquation; + ColorWriteTier colorWriteTier = DeviceCaps.SelectColorWriteTier(canEnable, canMask, + options.ColorWriteTier ?? DeviceCaps.FromEnvironment()); + + // Re-request only what the tier uses; the queries may have reported more. + colorWriteFeatures = new PhysicalDeviceColorWriteEnableFeaturesEXT + { + SType = StructureType.PhysicalDeviceColorWriteEnableFeaturesExt, + ColorWriteEnable = true, + }; + dynamicState3Features = new PhysicalDeviceExtendedDynamicState3FeaturesEXT + { + SType = StructureType.PhysicalDeviceExtendedDynamicState3FeaturesExt, + ExtendedDynamicState3ColorWriteMask = true, + ExtendedDynamicState3ColorBlendEnable = canBlend, + ExtendedDynamicState3ColorBlendEquation = canBlend, + }; + + // Optional: FAIL_ON_PIPELINE_COMPILE_REQUIRED for the background compile path + // (docs/vulkan.md#caches §2). Without it every pipeline compiles blocking. + var vulkan13Query = new PhysicalDeviceVulkan13Features { SType = StructureType.PhysicalDeviceVulkan13Features }; + var vulkan13QueryRoot = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &vulkan13Query, + }; + Api.GetPhysicalDeviceFeatures2(PhysicalDevice, &vulkan13QueryRoot); + bool pipelineCacheControl = vulkan13Query.PipelineCreationCacheControl; + + var vulkan13 = new PhysicalDeviceVulkan13Features + { + SType = StructureType.PhysicalDeviceVulkan13Features, + DynamicRendering = true, + Synchronization2 = true, + PipelineCreationCacheControl = pipelineCacheControl, + }; + var vulkan12 = new PhysicalDeviceVulkan12Features + { + SType = StructureType.PhysicalDeviceVulkan12Features, + PNext = &vulkan13, + ScalarBlockLayout = true, + TimelineSemaphore = true, + // Decision 9's bindless set: partially bound combined-image-sampler arrays + // written while bound. IsUsable already proved the device has them. + RuntimeDescriptorArray = true, + DescriptorBindingPartiallyBound = true, + DescriptorBindingSampledImageUpdateAfterBind = true, + }; + var features2 = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &vulkan12, + Features = enabledFeatures, + }; + + var deviceExtensions = new List(); + if (!options.Headless) deviceExtensions.Add("VK_KHR_swapchain"); + if (wantCheckpoints) deviceExtensions.Add("VK_NV_device_diagnostic_checkpoints"); + if (wantDeviceFault) deviceExtensions.Add("VK_EXT_device_fault"); + + // Optional tier: per-heap budgets from the driver; without it the + // allocator budgets heap x 0.7. The env override forces the fallback. + bool wantMemoryBudget = deviceExtensionsAvailable.ContainsKey("VK_EXT_memory_budget") + && Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_NO_MEMORY_BUDGET") != "1"; + if (wantMemoryBudget) deviceExtensions.Add("VK_EXT_memory_budget"); + if (colorWriteTier == ColorWriteTier.DynamicEnable) deviceExtensions.Add("VK_EXT_color_write_enable"); + if (colorWriteTier == ColorWriteTier.DynamicMask) deviceExtensions.Add("VK_EXT_extended_dynamic_state3"); + + // Optional feature structures stay on this stack until vkCreateDevice returns. + var presentId = new PhysicalDevicePresentIdFeaturesKHR + { + SType = StructureType.PhysicalDevicePresentIDFeaturesKhr, + }; + if (!options.Headless && deviceExtensionsAvailable.ContainsKey("VK_KHR_present_id")) + { + var query = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &presentId, + }; + Api.GetPhysicalDeviceFeatures2(PhysicalDevice, &query); + } + bool enablePresentId = presentId.PresentId; + if (enablePresentId) deviceExtensions.Add("VK_KHR_present_id"); + + string? maintenanceExtension = !options.Headless + && Array.IndexOf(EnabledInstanceExtensions, "VK_KHR_surface_maintenance1") >= 0 + && deviceExtensionsAvailable.ContainsKey("VK_KHR_swapchain_maintenance1") + ? "VK_KHR_swapchain_maintenance1" + : !options.Headless + && Array.IndexOf(EnabledInstanceExtensions, "VK_EXT_surface_maintenance1") >= 0 + && deviceExtensionsAvailable.ContainsKey("VK_EXT_swapchain_maintenance1") + ? "VK_EXT_swapchain_maintenance1" : null; + var maintenanceFeatures = new PhysicalDeviceSwapchainMaintenance1FeaturesEXT + { + SType = StructureType.PhysicalDeviceSwapchainMaintenance1FeaturesExt, + }; + if (maintenanceExtension != null) + { + var query = new PhysicalDeviceFeatures2 + { + SType = StructureType.PhysicalDeviceFeatures2, + PNext = &maintenanceFeatures, + }; + Api.GetPhysicalDeviceFeatures2(PhysicalDevice, &query); + } + bool enablePresentFences = maintenanceFeatures.SwapchainMaintenance1; + if (enablePresentFences) deviceExtensions.Add(maintenanceExtension!); + + void* optionalFeatures = null; + if (wantDeviceFault) { faultFeatures.PNext = optionalFeatures; optionalFeatures = &faultFeatures; } + if (colorWriteTier == ColorWriteTier.DynamicEnable) + { colorWriteFeatures.PNext = optionalFeatures; optionalFeatures = &colorWriteFeatures; } + if (colorWriteTier == ColorWriteTier.DynamicMask) + { dynamicState3Features.PNext = optionalFeatures; optionalFeatures = &dynamicState3Features; } + if (enablePresentId) { presentId.PNext = optionalFeatures; optionalFeatures = &presentId; } + if (enablePresentFences) { maintenanceFeatures.PNext = optionalFeatures; optionalFeatures = &maintenanceFeatures; } + vulkan13.PNext = optionalFeatures; + + nint extensionsPtr = deviceExtensions.Count > 0 + ? SilkMarshal.StringArrayToPtr(deviceExtensions) + : 0; + + try + { + var createInfo = new DeviceCreateInfo + { + SType = StructureType.DeviceCreateInfo, + PNext = &features2, + QueueCreateInfoCount = 1, + PQueueCreateInfos = &queueCreateInfo, + EnabledExtensionCount = (uint)deviceExtensions.Count, + PpEnabledExtensionNames = extensionsPtr == 0 ? null : (byte**)extensionsPtr, + }; + + Result result = Api.CreateDevice(PhysicalDevice, &createInfo, null, out Device device); + if (result != Result.Success) + { + failureReason = "vkCreateDevice failed: " + result; + return false; + } + Device = device; + } + finally + { + if (extensionsPtr != 0) SilkMarshal.Free(extensionsPtr); + } + + GraphicsQueue = Api.GetDeviceQueue(Device, family, 0); + LoadDiagnosticExtensions(wantCheckpoints, wantDeviceFault); + Capabilities = ReadCapabilities(); + Capabilities.ColorWriteTier = colorWriteTier; + Capabilities.PipelineCreationCacheControl = pipelineCacheControl; + Capabilities.ColorWriteEnable = colorWriteTier == ColorWriteTier.DynamicEnable; + Capabilities.DynamicColorWriteMask = colorWriteTier == ColorWriteTier.DynamicMask; + Capabilities.DynamicColorBlend = colorWriteTier == ColorWriteTier.DynamicMask && canBlend; + if (Capabilities.ColorWriteEnable && Api.TryGetDeviceExtension(Instance, Device, out ExtColorWriteEnable writeEnable)) + { + ColorWriteEnableApi = writeEnable; + } + if (Capabilities.DynamicColorWriteMask && Api.TryGetDeviceExtension(Instance, Device, out ExtExtendedDynamicState3 state3)) + { + DynamicState3Api = state3; + } + MemoryBudgetAvailable = wantMemoryBudget; + EnabledDeviceExtensions = deviceExtensions.ToArray(); + Capabilities.PresentIdEnabled = enablePresentId; + Capabilities.PresentFencesEnabled = enablePresentFences; + Allocator = new VulkanAllocator(this); + return true; + } + + /// Every device extension the driver advertises, with its revision. + private Dictionary EnumerateDeviceExtensions() + { + var names = new Dictionary(StringComparer.Ordinal); + + uint count = 0; + Result result = Api.EnumerateDeviceExtensionProperties(PhysicalDevice, (byte*)null, &count, null); + if (result != Result.Success || count == 0) return names; + + var properties = new ExtensionProperties[count]; + fixed (ExtensionProperties* propertiesPtr = properties) + { + Api.EnumerateDeviceExtensionProperties(PhysicalDevice, (byte*)null, &count, propertiesPtr); + + // The name is a fixed-size buffer, readable only through a pointer. + for (int i = 0; i < count; i++) + { + string? name = SilkMarshal.PtrToString((nint)propertiesPtr[i].ExtensionName); + if (name != null) names[name] = propertiesPtr[i].SpecVersion; + } + } + return names; + } + + private void LoadDiagnosticExtensions(bool checkpoints, bool deviceFault) + { + if (checkpoints) + { + nint set = (nint)Api.GetDeviceProcAddr(Device, "vkCmdSetCheckpointNV").Handle; + nint get = (nint)Api.GetDeviceProcAddr(Device, "vkGetQueueCheckpointDataNV").Handle; + if (set != 0 && get != 0) + { + _cmdSetCheckpoint = set; + _getQueueCheckpointData = get; + CheckpointsAvailable = true; + } + } + + if (deviceFault && Api.TryGetDeviceExtension(Instance, Device, out ExtDeviceFault fault)) + { + _deviceFault = fault; + DeviceFaultAvailable = true; + } + } + + /// Records a checkpoint marker into the command stream. No-op without the extension. + public void CmdSetCheckpoint(CommandBuffer commandBuffer, nint marker) + { + if (_cmdSetCheckpoint == 0) return; + ((delegate* unmanaged)_cmdSetCheckpoint)(commandBuffer, (void*)marker); + } + + /// + /// The last checkpoint each stage of the graphics queue reached. Meaningful + /// after a device loss. The caller synchronises the queue. + /// + public List<(PipelineStageFlags Stage, nint Marker)> ReadQueueCheckpoints() + { + var checkpoints = new List<(PipelineStageFlags, nint)>(); + if (_getQueueCheckpointData == 0) return checkpoints; + + var get = (delegate* unmanaged)_getQueueCheckpointData; + + uint count = 0; + get(GraphicsQueue, &count, null); + if (count == 0) return checkpoints; + + var data = new CheckpointDataNV[count]; + for (int i = 0; i < data.Length; i++) data[i].SType = StructureType.CheckpointDataNV; + fixed (CheckpointDataNV* dataPtr = data) + { + get(GraphicsQueue, &count, dataPtr); + } + + for (int i = 0; i < count; i++) + { + checkpoints.Add((data[i].Stage, (nint)data[i].PCheckpointMarker)); + } + return checkpoints; + } + + /// The driver's own account of a device loss, or null without the extension. + public string? ReadDeviceFault() + { + if (_deviceFault == null) return null; + + var counts = new DeviceFaultCountsEXT { SType = StructureType.DeviceFaultCountsExt }; + if (_deviceFault.GetDeviceFaultInfo(Device, &counts, null) != Result.Success) return null; + + var addresses = new DeviceFaultAddressInfoEXT[Math.Max(counts.AddressInfoCount, 1u)]; + var vendors = new DeviceFaultVendorInfoEXT[Math.Max(counts.VendorInfoCount, 1u)]; + var info = new DeviceFaultInfoEXT { SType = StructureType.DeviceFaultInfoExt }; + + // The binary blob is vendor-private and can be large; it is not asked for. + counts.VendorBinarySize = 0; + + var text = new System.Text.StringBuilder(); + + fixed (DeviceFaultAddressInfoEXT* addressPtr = addresses) + fixed (DeviceFaultVendorInfoEXT* vendorPtr = vendors) + { + info.PAddressInfos = counts.AddressInfoCount > 0 ? addressPtr : null; + info.PVendorInfos = counts.VendorInfoCount > 0 ? vendorPtr : null; + + Result result = _deviceFault.GetDeviceFaultInfo(Device, &counts, &info); + if (result != Result.Success && result != Result.Incomplete) return null; + + // The description strings are fixed-size buffers, readable only + // through a pointer, so everything is formatted while still pinned. + DeviceFaultInfoEXT* infoPtr = &info; + string description = SilkMarshal.PtrToString((nint)infoPtr->Description) ?? ""; + text.Append("Driver fault report: '").Append(description.Trim()).Append('\''); + + for (int i = 0; i < counts.AddressInfoCount; i++) + { + text.Append("; ").Append(addressPtr[i].AddressType) + .Append(" at 0x").Append(addressPtr[i].ReportedAddress.ToString("x")) + .Append(" (precision ").Append(addressPtr[i].AddressPrecision).Append(')'); + } + for (int i = 0; i < counts.VendorInfoCount; i++) + { + string vendor = SilkMarshal.PtrToString((nint)vendorPtr[i].Description) ?? ""; + text.Append("; vendor code ").Append(vendorPtr[i].VendorFaultCode) + .Append(" data ").Append(vendorPtr[i].VendorFaultData) + .Append(" '").Append(vendor.Trim()).Append('\''); + } + } + return text.ToString(); + } + + private readonly System.Collections.Concurrent.ConcurrentDictionary _formatFeatures = new(); + + /// The optimal-tiling features of on the selected device, cached. + public FormatFeatureFlags OptimalFormatFeatures(Format format) => + _formatFeatures.GetOrAdd(format, f => + { + FormatProperties properties; + Api.GetPhysicalDeviceFormatProperties(PhysicalDevice, f, &properties); + return properties.OptimalTilingFeatures; + }); + + private VulkanCapabilities ReadCapabilities() + { + PhysicalDeviceProperties properties = Api.GetPhysicalDeviceProperties(PhysicalDevice); + PhysicalDeviceFeatures features = Api.GetPhysicalDeviceFeatures(PhysicalDevice); + + var driverProperties = new PhysicalDeviceDriverProperties + { + SType = StructureType.PhysicalDeviceDriverProperties, + }; + var properties2 = new PhysicalDeviceProperties2 + { + SType = StructureType.PhysicalDeviceProperties2, + PNext = &driverProperties, + }; + Api.GetPhysicalDeviceProperties2(PhysicalDevice, &properties2); + + return new VulkanCapabilities + { + DeviceName = SilkMarshal.PtrToString((nint)properties.DeviceName) ?? "unknown", + DriverName = SilkMarshal.PtrToString((nint)driverProperties.DriverName) ?? "unknown", + VendorId = properties.VendorID, + DeviceId = properties.DeviceID, + DriverVersion = properties.DriverVersion, + PipelineCacheUuid = new ReadOnlySpan(properties.PipelineCacheUuid, 16).ToArray(), + ApiVersion = properties.ApiVersion, + DeviceType = properties.DeviceType, + MaxImageDimension2D = properties.Limits.MaxImageDimension2D, + WideLines = features.WideLines, + LineWidthMin = properties.Limits.LineWidthRange[0], + LineWidthMax = properties.Limits.LineWidthRange[1], + FillModeNonSolid = features.FillModeNonSolid, + SamplerAnisotropy = features.SamplerAnisotropy, + MultiDrawIndirect = features.MultiDrawIndirect, + OcclusionQueryPrecise = features.OcclusionQueryPrecise, + MaxSamplerLodBias = properties.Limits.MaxSamplerLodBias, + MaxBoundDescriptorSets = (int)properties.Limits.MaxBoundDescriptorSets, + MinUniformBufferOffsetAlignment = properties.Limits.MinUniformBufferOffsetAlignment, + MinStorageBufferOffsetAlignment = properties.Limits.MinStorageBufferOffsetAlignment, + MaxUniformBufferRange = properties.Limits.MaxUniformBufferRange, + MaxColorAttachments = properties.Limits.MaxColorAttachments, + DescriptorIndexing = ReadDescriptorIndexingSupport(PhysicalDevice), + }; + } + + public static string VersionString(uint version) => + $"{version >> 22}.{(version >> 12) & 0x3FF}.{version & 0xFFF}"; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + if (Device.Handle != 0) + { + VulkanStats.WaitDeviceIdle(Api, Device); + + // Memory blocks are freed while the device still exists, and after + // the wait, so nothing is executing against them. + Allocator?.Dispose(); + Api.DestroyDevice(Device, null); + } + + if (_debugUtils != null && _debugMessenger.Handle != 0) + { + _debugUtils.DestroyDebugUtilsMessenger(Instance, _debugMessenger, null); + _debugUtils.Dispose(); + } + + if (Instance.Handle != 0) + { + Api.DestroyInstance(Instance, null); + } + + Api?.Dispose(); + } +} diff --git a/Optimum.Render.Vulkan/Core/VulkanResources.cs b/Optimum.Render.Vulkan/Core/VulkanResources.cs new file mode 100644 index 00000000..9d8534bf --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VulkanResources.cs @@ -0,0 +1,437 @@ +using System; +using System.Threading; +using Silk.NET.Vulkan; +using Buffer = Silk.NET.Vulkan.Buffer; +using System.Collections.Generic; + +namespace Optimum.Render.Vulkan.Core; + +// Silk.NET.Vulkan.Buffer collides with System.Buffer. + +/// +/// Process-unique ids for device resources. +/// +/// A Vulkan handle identifies an object only while it lives: destroy an image +/// view and the driver is free to hand the very same handle value to the next +/// one created. Anything that remembers a resource by handle - the descriptor +/// set cache does - would then mistake the newcomer for the dead one and serve +/// a set that points at freed memory. An id that is never reused is what such +/// a cache has to key on instead. +/// +internal static class ResourceIds +{ + private static long _next; + + public static ulong Next() => (ulong)Interlocked.Increment(ref _next); + + /// The highest id issued so far (0 before the first). + public static ulong Highest => (ulong)Interlocked.Read(ref _next); +} + +/// A device buffer with its backing memory. +internal sealed unsafe class VulkanBuffer : IDisposable +{ + private readonly VulkanContext _context; + private bool _disposed; + + public Buffer Handle { get; } + public ulong Size { get; } + + /// What the buffer was created for; the barriers around a staged copy name these uses. + public BufferUsageFlags Usage { get; } + + /// Never reused, unlike ; see . + public ulong Id { get; } = ResourceIds.Next(); + + private MemoryAllocation _allocation; + + /// Which block this buffer's memory came from. For tests. + internal ulong MemoryHandleForTest => _allocation.Memory.Handle; + + /// The block, offset and size this buffer occupies. For tests. + internal MemoryAllocation Allocation => _allocation; + + /// Non-zero when the allocation is host visible and mapped. + public IntPtr Mapped { get; private set; } + + /// + /// The frame command buffer generation that last used this buffer; see + /// . + /// + internal long FrameUse; + + public VulkanBuffer(VulkanContext context, ulong size, BufferUsageFlags usage, MemoryPropertyFlags properties) + : this(context, size, usage, properties, VulkanAllocator.InferClass(properties, linear: true)) + { + } + + /// A buffer in an explicit pool class; see . + public VulkanBuffer(VulkanContext context, ulong size, BufferUsageFlags usage, MemoryPropertyFlags properties, + MemoryPoolClass poolClass) + { + _context = context; + Size = size; + Usage = usage; + + var createInfo = new BufferCreateInfo + { + SType = StructureType.BufferCreateInfo, + Size = size, + Usage = usage, + SharingMode = SharingMode.Exclusive, + }; + + Vk api = context.Api; + if (api.CreateBuffer(context.Device, &createInfo, null, out Buffer buffer) != Result.Success) + { + throw new InvalidOperationException("vkCreateBuffer failed"); + } + Handle = buffer; + + MemoryRequirements requirements = VulkanAllocator.BufferRequirements(context, buffer, out bool dedicated); + + // A buffer is linear, so it shares blocks only with other buffers. + _allocation = context.Allocator.Allocate( + requirements, properties, linear: true, $"a {size} byte buffer", poolClass, dedicated, buffer, default); + + api.BindBufferMemory(context.Device, buffer, _allocation.Memory, _allocation.Offset); + Mapped = _allocation.Mapped; + if (context.PoisonFreshResources && Mapped != IntPtr.Zero) + { + VulkanPoison.FillHostMemory(Mapped, size); + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + // The mapping belongs to the block, not to this buffer, so it is not + // unmapped here - the region simply goes back to the pool. + Mapped = IntPtr.Zero; + _context.Api.DestroyBuffer(_context.Device, Handle, null); + _context.Allocator.Free(_allocation); + } +} + +/// An image, its memory and a default view. +internal sealed unsafe class VulkanImage : IDisposable +{ + private readonly VulkanContext _context; + private bool _disposed; + + public Image Handle { get; } + public ImageView View { get; } + + private MemoryAllocation _allocation; + public Format Format { get; } + public uint Width { get; } + public uint Height { get; } + + /// + /// Tracked so transitions can name the right old layout. Vulkan has no way to + /// query it, so the backend has to remember. + /// + public ImageLayout Layout { get; set; } = ImageLayout.Undefined; + + public VulkanImage( + VulkanContext context, uint width, uint height, Format format, + ImageUsageFlags usage, ImageAspectFlags aspect) + { + _context = context; + Width = width; + Height = height; + Format = format; + + var createInfo = new ImageCreateInfo + { + SType = StructureType.ImageCreateInfo, + ImageType = ImageType.Type2D, + Format = format, + Extent = new Extent3D(width, height, 1), + MipLevels = 1, + ArrayLayers = 1, + Samples = SampleCountFlags.Count1Bit, + Tiling = ImageTiling.Optimal, + Usage = usage, + SharingMode = SharingMode.Exclusive, + InitialLayout = ImageLayout.Undefined, + }; + + Vk api = context.Api; + if (api.CreateImage(context.Device, &createInfo, null, out Image image) != Result.Success) + { + throw new InvalidOperationException("vkCreateImage failed"); + } + Handle = image; + + MemoryRequirements requirements = VulkanAllocator.ImageRequirements(context, image, out bool dedicated); + + // Optimally tiled, so it never shares a block with a buffer. + _allocation = context.Allocator.Allocate( + requirements, MemoryPropertyFlags.DeviceLocalBit, linear: false, "an image", + MemoryPoolClass.DeviceImages, dedicated, default, image); + api.BindImageMemory(context.Device, image, _allocation.Memory, _allocation.Offset); + + var viewInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = image, + ViewType = ImageViewType.Type2D, + Format = format, + SubresourceRange = new ImageSubresourceRange(aspect, 0, 1, 0, 1), + }; + if (api.CreateImageView(context.Device, &viewInfo, null, out ImageView view) != Result.Success) + { + throw new InvalidOperationException("vkCreateImageView failed"); + } + View = view; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + Vk api = _context.Api; + api.DestroyImageView(_context.Device, View, null); + api.DestroyImage(_context.Device, Handle, null); + _context.Allocator.Free(_allocation); + } +} + +/// +/// Turns a Vulkan result into a failure that says something. +/// +/// The backend used to ignore every result it got back. A lost device then +/// looked like nothing at all from inside: submits kept "succeeding", fence +/// waits returned immediately, and the client spun at a few frames a second +/// forever with no error anywhere - the fault was only visible to an external +/// overlay. A hang with no message is the worst possible failure mode, so every +/// submit, wait, acquire and present is checked, and a device loss is reported +/// where it happens rather than inferred later. +/// +internal static class VulkanResult +{ + /// Set once the device is gone, so the failure is reported once and not per call. + public static volatile bool DeviceLost; + + /// Called with a description the moment something fails. + public static Action? OnFailure; + + /// + /// Asked to describe a device loss after the fact, so the message can say + /// what the GPU was doing rather than only that it stopped. Null when + /// nothing on the device can answer. + /// + public static Func? DescribeDeviceLoss; + + public static void Check(Result result, string operation) + { + if (result == Result.Success || result == Result.SuboptimalKhr) return; + + bool lost = result is Result.ErrorDeviceLost; + if (lost && DeviceLost) return; + if (lost) DeviceLost = true; + + string message = lost + ? operation + " reported the device was lost. The GPU driver aborted the work this " + + "backend submitted; the session cannot continue." + : operation + " failed with " + result; + + if (lost) + { + string? detail; + try + { + detail = DescribeDeviceLoss?.Invoke(); + } + catch (Exception e) + { + detail = "Describing the loss itself failed: " + e.Message; + } + if (!string.IsNullOrEmpty(detail)) message += " " + detail; + message += " (" + VulkanMemory.LiveAllocations + " live device allocations.)"; + } + + OnFailure?.Invoke(message); + throw new InvalidOperationException(message); + } +} + +internal static unsafe class VulkanMemory +{ + /// + /// How many device allocations are currently outstanding. + /// + /// Vulkan caps this per device - commonly 4096 - and every buffer and image + /// here owns its own allocation, so a world with a few hundred chunk meshes + /// approaches the limit fast. Past it vkAllocateMemory starts failing, and an + /// unchecked failure binds a null handle and faults the GPU rather than + /// reporting anything. Counted so the failure can name its cause. + /// + private static int _liveAllocations; + + public static int LiveAllocations => Volatile.Read(ref _liveAllocations); + + public static void NoteAllocation() => Interlocked.Increment(ref _liveAllocations); + + public static void NoteFree() => Interlocked.Decrement(ref _liveAllocations); + + /// + /// Allocates device memory, failing with a message that says what ran out. + /// + public static DeviceMemory Allocate(VulkanContext context, MemoryAllocateInfo allocateInfo, string what) + { + Result result = context.Api.AllocateMemory(context.Device, &allocateInfo, null, out DeviceMemory memory); + if (result != Result.Success) + { + throw new InvalidOperationException( + $"vkAllocateMemory failed for {what} with {result} after {LiveAllocations} live allocations " + + $"({allocateInfo.AllocationSize} bytes requested)"); + } + + NoteAllocation(); + VulkanStats.NoteAllocation(); + return memory; + } + + /// + /// Picks a memory type satisfying both the resource's type mask and the + /// requested properties. + /// + public static uint FindMemoryType(VulkanContext context, uint typeBits, MemoryPropertyFlags properties) + { + context.Api.GetPhysicalDeviceMemoryProperties(context.PhysicalDevice, out PhysicalDeviceMemoryProperties memory); + + for (uint i = 0; i < memory.MemoryTypeCount; i++) + { + bool typeAllowed = (typeBits & (1u << (int)i)) != 0; + if (!typeAllowed) continue; + + MemoryPropertyFlags flags = memory.MemoryTypes[(int)i].PropertyFlags; + if ((flags & properties) == properties) return i; + } + + throw new InvalidOperationException($"no memory type with {properties}"); + } +} + +/// +/// Which format a storage image is created in. A compute pass names the format it +/// wants (the AO working term wants R8_UNORM, the prefiltered depth R32F); a device +/// that cannot use that format as a storage image and sample it gets the first +/// wider format of the same kind that it can, ending in RGBA8 for unsigned +/// normalised formats and RGBA32F for float ones (the formats Vulkan guarantees +/// storage support for). A shader writing the first channel of the wider format +/// reads the same value back, so a fallback costs memory, never correctness. +/// +/// Pure: the feature lookup is passed in, so the choice is testable without a device. +/// +internal static class StorageFormats +{ + /// What a storage image must support: storage writes and sampling. + public const FormatFeatureFlags Required = FormatFeatureFlags.StorageImageBit | FormatFeatureFlags.SampledImageBit; + + /// The candidates for , the requested format first. + public static IReadOnlyList CandidatesFor(Format requested) => requested switch + { + Format.R8Unorm => new[] { Format.R8Unorm, Format.R8G8Unorm, Format.R8G8B8A8Unorm }, + Format.R8G8Unorm => new[] { Format.R8G8Unorm, Format.R8G8B8A8Unorm }, + Format.R16Unorm => new[] { Format.R16Unorm, Format.R16G16B16A16Unorm, Format.R8G8B8A8Unorm }, + Format.R16Sfloat => new[] { Format.R16Sfloat, Format.R32Sfloat, Format.R16G16B16A16Sfloat, Format.R32G32B32A32Sfloat }, + Format.R32Sfloat => new[] { Format.R32Sfloat, Format.R32G32B32A32Sfloat }, + Format.R16G16Sfloat => new[] { Format.R16G16Sfloat, Format.R16G16B16A16Sfloat, Format.R32G32B32A32Sfloat }, + Format.R16G16B16A16Sfloat => new[] { Format.R16G16B16A16Sfloat, Format.R32G32B32A32Sfloat }, + Format.R8G8B8A8Unorm => new[] { Format.R8G8B8A8Unorm }, + _ => new[] { requested, Format.R8G8B8A8Unorm }, + }; + + /// + /// The first candidate whose optimal-tiling features include ; + /// RGBA8 when none does (every Vulkan device supports it as a storage image). + /// + public static Format Choose(Format requested, Func optimalFeatures) + { + foreach (Format candidate in CandidatesFor(requested)) + { + if ((optimalFeatures(candidate) & Required) == Required) return candidate; + } + return Format.R8G8B8A8Unorm; + } + + /// Whether a colour attachment usage may be added: the format must support it. + public static bool SupportsColorAttachment(FormatFeatureFlags features) => + (features & FormatFeatureFlags.ColorAttachmentBit) != 0; +} + +/// +/// The values poison mode (OPTIMUM_VULKAN_POISON=1) writes into fresh resources. +/// +/// OpenGL and Vulkan both leave new storage undefined, but in practice GL +/// drivers hand out zeroed memory and Vulkan allocators hand out whatever the +/// previous tenant left. A read of never-written content therefore "works" on +/// one backend and flickers on the other. Poison makes such a read loud and +/// identical every frame: NaN for float formats, magenta (alpha 1) for +/// normalised and sRGB colour, 0xDEADBEEF for integer formats and host memory, +/// 0.5 for depth. +/// +internal static unsafe class VulkanPoison +{ + public const uint Word = 0xDEADBEEF; + public const float Depth = 0.5f; + + public static bool IsCompressed(Format format) => + format.ToString().Contains("Block", StringComparison.Ordinal); + + public static bool IsFloat(Format format) + { + string name = format.ToString(); + return name.Contains("Sfloat", StringComparison.Ordinal) || name.Contains("Ufloat", StringComparison.Ordinal); + } + + public static bool IsInteger(Format format) + { + string name = format.ToString(); + return name.Contains("Uint", StringComparison.Ordinal) || name.Contains("Sint", StringComparison.Ordinal); + } + + public static ClearColorValue ColorFor(Format format) + { + var value = new ClearColorValue(); + if (IsFloat(format)) + { + value.Float32_0 = float.NaN; + value.Float32_1 = float.NaN; + value.Float32_2 = float.NaN; + value.Float32_3 = float.NaN; + } + else if (IsInteger(format)) + { + // Uint and Sint clears read the same union bits. + value.Uint32_0 = Word; + value.Uint32_1 = Word; + value.Uint32_2 = Word; + value.Uint32_3 = Word; + } + else + { + value.Float32_0 = 1f; + value.Float32_1 = 0f; + value.Float32_2 = 1f; + value.Float32_3 = 1f; + } + return value; + } + + /// Writes 0xDEADBEEF as little-endian words over the whole range, a partial word at the tail. + public static void FillHostMemory(IntPtr memory, ulong size) + { + byte* bytes = (byte*)memory; + ulong words = size / 4; + uint* wordPointer = (uint*)bytes; + for (ulong i = 0; i < words; i++) wordPointer[i] = Word; + for (ulong i = words * 4; i < size; i++) bytes[i] = (byte)(Word >> (int)(8 * (i % 4))); + } +} diff --git a/Optimum.Render.Vulkan/Core/VulkanStats.cs b/Optimum.Render.Vulkan/Core/VulkanStats.cs new file mode 100644 index 00000000..654c9e52 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/VulkanStats.cs @@ -0,0 +1,978 @@ +using System; +using System.Diagnostics; +using System.Globalization; +using System.Text; +using System.Threading; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Every place the backend makes the CPU wait on the GPU or the presentation +/// engine. Fixed and small so the counters are two flat arrays; the order is the +/// order of the tokens on the stats.waits line. +/// +internal enum WaitSite +{ + /// + /// vkWaitSemaphores on the Frame timeline for value n - FramesInFlight at the + /// start of frame n; exactly one per frame start, the ring's only steady-state wait. + /// + FramePacing = 0, + /// + /// An upload that waited for the GPU. Retired in Phase 1B step 3: uploads ride + /// the next frame submission (UploadManager) and never wait, so this stays + /// zero; the token stays for log compatibility and the pacing gate. + /// + UploadSubmit = 1, + /// + /// The Frame timeline wait inside a mid-frame flush. Retired in Phase 1B: + /// readbacks and uploads submit partially without waiting and queries never + /// flush, so this stays zero; the token stays for log compatibility. + /// + FlushFrame = 2, + /// vkDeviceWaitIdle, wherever it is called. + DeviceWaitIdle = 3, + /// + /// A readback the caller needs now: the Frame timeline value of the partial + /// submission that carried the copy, or a between-frames setup fence. + /// + Readback = 4, + /// + /// Polling an occlusion query until its result is available. Retired in + /// Phase 1B (QueryRing reads results without waiting); stays zero. + /// + OcclusionQuery = 5, + /// vkAcquireNextImageKHR. + SwapchainAcquire = 6, + /// vkQueuePresentKHR, including the queue lock. + Present = 7, + /// + /// vkQueueSubmit of a frame slot, including the queue lock. A worker's + /// synchronous upload holds that lock through its fence wait, so the render + /// thread can stall here on GPU work it did not issue. + /// + QueueSubmit = 8, + +} + +/// +/// Per-second counters for the work the backend does, so a slow phase can be +/// attributed rather than guessed at. +/// +/// A frame time on its own says the renderer is slow; it does not say whether +/// the cost is device allocations, blocking uploads waiting on the GPU, or +/// descriptor churn. These count each of those and report them together. +/// +/// Off unless OPTIMUM_VULKAN_STATS names a file. The counters themselves are +/// always live - they are interlocked increments on paths that already cost +/// microseconds apiece, so they do not need gating, and having them unconditional +/// means a report can be turned on for a session that is already misbehaving. +/// +/// Each sample starts with a human-readable summary. The remaining lines +/// carry key=value counters and timings for scripts +/// (scripts/dev/pacing-gate.sh): +/// +/// stats 1.0s: 60 frames (16.7 ms/frame), ... +/// stats.pacing samples=512 p50_ms=16.667 p95_ms=17.100 p99_ms=18.300 stddev_ms=0.420 stutters=0 +/// stats.waits frame_pacing_n=60 frame_pacing_ms=812.4 upload_submit_n=0 upload_submit_ms=0.0 ... queue_submit_n=60 queue_submit_ms=1.9 +/// stats.counters blocking_uploads=0 uploads=0 scopes=900 barriers=12 rebar_fallbacks=0 dynamic_state=12600 uniform_ring_used=412800 uniform_ring_capacity=16777216 +/// stats.latency frames=60 input_mean_ms=0.02 input_p99_ms=0.04 ... total_mean_ms=16.60 total_p99_ms=18.20 +/// +/// +internal static class VulkanStats +{ + /// Token stems of , indexed by its value. + public static readonly string[] WaitSiteTokens = + { + "frame_pacing", + "upload_submit", + "flush_frame", + "device_wait_idle", + "readback", + "occlusion_query", + "swapchain_acquire", + "present", + "queue_submit", + }; + + public const int WaitSiteCount = 9; + + /// + /// Dynamic-state commands VulkanDevice.ApplyDynamicState can record for + /// one draw: all of them, at the first draw of a command buffer. Later draws + /// record only the ones whose value changed (Phase 1B step 6). A source test + /// keeps this equal to the calls in that method. + /// + public const int DynamicStateCommandsPerDraw = 14; + + private static long _allocations; + private static long _uploads; + private static long _uploadWaitTicks; + private static long _texturesCreated; + private static long _texturesDeleted; + private static long _frames; + private static long _droppedMeshWrites; + private static long _uniformOverflows; + + private static long _blockingUploads; + private static long _uploadRequests; + private static long _scopesOpened; + private static long _imageBarriers; + private static long _barrierCommands; + private static long _rebarFallbacks; + private static long _dynamicStateCommands; + private static long _uniformRingPeak; + private static long _uniformRingCapacity; + private static readonly long[] _waitCounts = new long[WaitSiteCount]; + private static readonly long[] _waitTicks = new long[WaitSiteCount]; + + /// CPU frame intervals of the last 512 frames, any device. + public static readonly FrameIntervalRing FrameIntervals = new(FrameIntervalRing.DefaultCapacity); + + /// A mesh write that could not land; see MeshManager.Write. + public static void NoteDroppedMeshWrite() => Interlocked.Increment(ref _droppedMeshWrites); + + public static long DroppedMeshWrites => Interlocked.Read(ref _droppedMeshWrites); + + /// A named uniform block that did not fit the frame ring and took a transient buffer instead. + public static void NoteUniformOverflow() => Interlocked.Increment(ref _uniformOverflows); + + public static long UniformOverflows => Interlocked.Read(ref _uniformOverflows); + + public static void NoteAllocation() => Interlocked.Increment(ref _allocations); + public static void NoteTextureCreated() => Interlocked.Increment(ref _texturesCreated); + public static void NoteTextureDeleted() => Interlocked.Increment(ref _texturesDeleted); + public static void NoteFrame() => Interlocked.Increment(ref _frames); + + /// Textures deleted since the last . + public static long TexturesDeleted => Interlocked.Read(ref _texturesDeleted); + + /// + /// A synchronous setup submission of any kind (uploads and readbacks alike). + /// Feeds the original line's "blocking uploads" figure, whose meaning is kept. + /// + public static void NoteUpload(long elapsedTicks) + { + Interlocked.Increment(ref _uploads); + Interlocked.Add(ref _uploadWaitTicks, elapsedTicks); + } + + /// A texture upload or mip generation was requested, whether or not it waited. + public static void NoteUploadRequest() => Interlocked.Increment(ref _uploadRequests); + + /// An upload that really waited on a fence or the queue lock. + public static void NoteBlockingUpload() => Interlocked.Increment(ref _blockingUploads); + + public static long BlockingUploads => Interlocked.Read(ref _blockingUploads); + public static long UploadRequests => Interlocked.Read(ref _uploadRequests); + + private static long _inlineUploads; + private static long _stagingOverflows; + private static long _uploadBatchGrowths; + + /// + /// An upload recorded into the frame command buffer because that command + /// buffer already used its destination (GL order), instead of the upload batch. + /// + public static void NoteInlineUpload() => Interlocked.Increment(ref _inlineUploads); + + /// An upload that did not fit its batch's staging region and took a dedicated staging buffer. + public static void NoteStagingOverflow() => Interlocked.Increment(ref _stagingOverflows); + + /// An upload batch created beyond the staging ring's regions (more batches in flight than frames). + public static void NoteUploadBatchGrowth() => Interlocked.Increment(ref _uploadBatchGrowths); + + public static long InlineUploads => Interlocked.Read(ref _inlineUploads); + public static long StagingOverflows => Interlocked.Read(ref _stagingOverflows); + public static long UploadBatchGrowths => Interlocked.Read(ref _uploadBatchGrowths); + + /// One vkCmdBeginRendering. + public static void NoteScopeOpened() => Interlocked.Increment(ref _scopesOpened); + + public static long ScopesOpened => Interlocked.Read(ref _scopesOpened); + + private static long _maskRestarts; + private static long _feedbackSplits; + + /// + /// A scope restart that reopened exactly the attachment set it closed (same + /// views, same layouts). Draw-buffer and colour-mask changes only alter write + /// masks (Phase 2, C4), so this must stay 0. + /// + public static void NoteMaskRestart() => Interlocked.Increment(ref _maskRestarts); + + /// + /// A scope restart because a draw samples a bound colour attachment its draw + /// buffers exclude (the composition pass reads Primary 1), or because such a + /// slot rejoins the scope once its draw buffer is enabled again. + /// + public static void NoteFeedbackSplit() => Interlocked.Increment(ref _feedbackSplits); + + private static long _passes; + private static long _planHits; + private static long _planMisses; + private static long _inPassClears; + private static long _promotedClears; + private static long _standaloneClears; + private static long _passSplits; + + /// A frame-graph pass opened its scope (a declared pass, or a scope no declaration covered). + public static void NotePass() => Interlocked.Increment(ref _passes); + + /// A frame whose passes matched the plan solved from the previous frame exactly. + public static void NotePlanHit() => Interlocked.Increment(ref _planHits); + + /// A frame recorded conservatively because it did not match the plan. + public static void NotePlanMiss() => Interlocked.Increment(ref _planMisses); + + /// A clear recorded as vkCmdClearAttachments inside an open pass. + public static void NoteInPassClear() => Interlocked.Increment(ref _inPassClears); + + /// A clear issued with no pass open that became LOAD_OP_CLEAR. + public static void NotePromotedClear() => Interlocked.Increment(ref _promotedClears); + + /// A promoted clear recorded as a clear-image command (its image was used before a pass attached it). + public static void NoteStandaloneClear() => Interlocked.Increment(ref _standaloneClears); + + /// A second rendering scope inside one declared pass. + public static void NotePassSplit() => Interlocked.Increment(ref _passSplits); + + private static long _computePasses; + private static long _dispatches; + + /// A compute pass recorded (its barriers, pipeline and set), outside any rendering scope. + public static void NoteComputePass() => Interlocked.Increment(ref _computePasses); + + /// One vkCmdDispatch. + public static void NoteDispatch() => Interlocked.Increment(ref _dispatches); + + private static long _transientBytes; + private static long _aliasedBytesPeak; + private static long _transientLeases; + private static long _aliasedLeases; + private static long _readSelfCopies; + private static long _readSelfPool; + + /// + /// One finished frame's transients (Phase 2 step 4): bytes of transient images + /// (the opted-in textures plus the allocator's physical images), bytes of leases + /// served by an image an earlier lease of the frame used, the lease counts, and the + /// ReadSelf copies the pool holds. + /// + public static void NoteTransientFrame(ulong transientBytes, ulong aliasedBytes, int leases, int aliasedLeases, + int readSelfPool) + { + Interlocked.Exchange(ref _transientBytes, (long)Math.Min(transientBytes, long.MaxValue)); + long aliased = (long)Math.Min(aliasedBytes, long.MaxValue); + long peak = Interlocked.Read(ref _aliasedBytesPeak); + while (aliased > peak) + { + long seen = Interlocked.CompareExchange(ref _aliasedBytesPeak, aliased, peak); + if (seen == peak) break; + peak = seen; + } + Interlocked.Add(ref _transientLeases, leases); + Interlocked.Add(ref _aliasedLeases, aliasedLeases); + Interlocked.Exchange(ref _readSelfPool, readSelfPool); + } + + /// A draw sampled a colour attachment it writes and took a pooled ReadSelf copy. + public static void NoteReadSelfCopy() => Interlocked.Increment(ref _readSelfCopies); + + public static long ReadSelfCopies => Interlocked.Read(ref _readSelfCopies); + + public static long MaskRestarts => Interlocked.Read(ref _maskRestarts); + public static long FeedbackSplits => Interlocked.Read(ref _feedbackSplits); + + /// Image memory barriers recorded into a command buffer. + public static void NoteImageBarriers(int count) => Interlocked.Add(ref _imageBarriers, count); + + public static long ImageBarriers => Interlocked.Read(ref _imageBarriers); + + /// One vkCmdPipelineBarrier2 carrying image barriers (a BarrierBatcher flush). + public static void NoteBarrierCommand() => Interlocked.Increment(ref _barrierCommands); + + public static long BarrierCommands => Interlocked.Read(ref _barrierCommands); + + /// A buffer that asked for ReBAR (device-local and host-visible) and fell back to plain host memory. + public static void NoteRebarFallback() => Interlocked.Increment(ref _rebarFallbacks); + + public static long RebarFallbacks => Interlocked.Read(ref _rebarFallbacks); + + private static long _bindlessWrites; + private static long _bindlessFlushes; + private static long _bindlessPlaceholderResolutions; + + /// + /// One vkUpdateDescriptorSets on the bindless set carrying + /// slot writes (new slots and placeholders written back into freed ones). + /// + public static void NoteBindlessFlush(int writes) + { + Interlocked.Increment(ref _bindlessFlushes); + Interlocked.Add(ref _bindlessWrites, writes); + } + + /// A bindless lookup that resolved to a placeholder slot: no texture, a texture of the wrong kind, or a full array. + public static void NoteBindlessPlaceholderResolution() + { + Interlocked.Increment(ref _bindlessPlaceholderResolutions); + Interlocked.Increment(ref _intervalBindlessPlaceholders); + } + + // Cumulative (tests read them) and per sample interval (the counters line). + private static long _pushConstantWrites; + private static long _storageSetBinds; + private static long _bindlessSlotResolutions; + private static long _samplerPlaceholders; + private static long _intervalPushConstantWrites; + private static long _intervalStorageSetBinds; + private static long _intervalBindlessSlots; + private static long _intervalBindlessPlaceholders; + + /// A draw whose slot indices differed from what its recording last received: one vkCmdPushConstants. + public static void NotePushConstantWrite() + { + Interlocked.Increment(ref _pushConstantWrites); + Interlocked.Increment(ref _intervalPushConstantWrites); + } + + /// A draw that bound set 2 because its set or its record's dynamic offset changed. + public static void NoteStorageSetBind() + { + Interlocked.Increment(ref _storageSetBinds); + Interlocked.Increment(ref _intervalStorageSetBinds); + } + + /// A draw's sampler resolved through the bindless table (placeholder slots included). + public static void NoteBindlessSlotResolution() + { + Interlocked.Increment(ref _bindlessSlotResolutions); + Interlocked.Increment(ref _intervalBindlessSlots); + } + + private static long _nativePasses; + private static long _nativeDraws; + private static long _intervalNativePasses; + private static long _intervalNativeDraws; + + /// A pass a native render system declared with explicit writes and reads (no draw-buffer mask). + public static void NoteNativePass() + { + Interlocked.Increment(ref _nativePasses); + Interlocked.Increment(ref _intervalNativePasses); + } + + /// A draw a native render system recorded through its own pipeline, without the GL state tracker. + public static void NoteNativeDraw() + { + Interlocked.Increment(ref _nativeDraws); + Interlocked.Increment(ref _intervalNativeDraws); + } + + public static long NativePasses => Interlocked.Read(ref _nativePasses); + public static long NativeDraws => Interlocked.Read(ref _nativeDraws); + + private static long _nativeFullscreenDraws; + private static long _nativeMeshDraws; + private static long _nativeInstancedDraws; + private static long _nativeIndirectDraws; + private static long _intervalNativeFullscreenDraws; + private static long _intervalNativeMeshDraws; + private static long _intervalNativeInstancedDraws; + private static long _intervalNativeIndirectDraws; + + /// A native draw of the fullscreen triangle: a post or TAA chain pass. + public static void NoteNativeFullscreenDraw() + { + Interlocked.Increment(ref _nativeFullscreenDraws); + Interlocked.Increment(ref _intervalNativeFullscreenDraws); + } + + /// A native draw of one mesh (indexed or not), one instance: sky, entities, GUI quads. + public static void NoteNativeMeshDraw() + { + Interlocked.Increment(ref _nativeMeshDraws); + Interlocked.Increment(ref _intervalNativeMeshDraws); + } + + /// A native draw of one mesh with more than one instance: the particle pools. + public static void NoteNativeInstancedDraw() + { + Interlocked.Increment(ref _nativeInstancedDraws); + Interlocked.Increment(ref _intervalNativeInstancedDraws); + } + + /// A native indirect multi-draw out of the per-slot indirect ring: chunk pools and decals. + public static void NoteNativeIndirectDraw() + { + Interlocked.Increment(ref _nativeIndirectDraws); + Interlocked.Increment(ref _intervalNativeIndirectDraws); + } + + public static long NativeFullscreenDraws => Interlocked.Read(ref _nativeFullscreenDraws); + public static long NativeMeshDraws => Interlocked.Read(ref _nativeMeshDraws); + public static long NativeInstancedDraws => Interlocked.Read(ref _nativeInstancedDraws); + public static long NativeIndirectDraws => Interlocked.Read(ref _nativeIndirectDraws); + + /// A draw's frame texture (set 0) resolved to its placeholder: nothing suitable bound. + public static void NoteSamplerPlaceholder() + { + Interlocked.Increment(ref _samplerPlaceholders); + Interlocked.Increment(ref _intervalBindlessPlaceholders); + } + + public static long PushConstantWrites => Interlocked.Read(ref _pushConstantWrites); + public static long StorageSetBinds => Interlocked.Read(ref _storageSetBinds); + public static long BindlessSlotResolutions => Interlocked.Read(ref _bindlessSlotResolutions); + public static long SamplerPlaceholders => Interlocked.Read(ref _samplerPlaceholders); + + public static long BindlessWrites => Interlocked.Read(ref _bindlessWrites); + public static long BindlessFlushes => Interlocked.Read(ref _bindlessFlushes); + public static long BindlessPlaceholderResolutions => Interlocked.Read(ref _bindlessPlaceholderResolutions); + + /// A multi-draw that did not fit its frame slot's indirect buffer and took an overflow buffer. + public static void NoteIndirectOverflow() => Interlocked.Increment(ref _indirectOverflows); + + public static long IndirectOverflows => Interlocked.Read(ref _indirectOverflows); + + private static long _indirectOverflows; + + public static void NoteDynamicStateCommands(int count) => Interlocked.Add(ref _dynamicStateCommands, count); + + public static long DynamicStateCommands => Interlocked.Read(ref _dynamicStateCommands); + + /// + /// Uniform-ring bytes one frame slot used by the time it was submitted, and + /// that slot's capacity. The sample reports the peak since the last sample. + /// + public static void NoteUniformRingUse(ulong used, ulong capacity) + { + long value = (long)Math.Min(used, long.MaxValue); + long peak = Interlocked.Read(ref _uniformRingPeak); + while (value > peak) + { + long seen = Interlocked.CompareExchange(ref _uniformRingPeak, value, peak); + if (seen == peak) break; + peak = seen; + } + Interlocked.Exchange(ref _uniformRingCapacity, (long)Math.Min(capacity, long.MaxValue)); + } + + public static long UniformRingPeak => Interlocked.Read(ref _uniformRingPeak); + + /// A timestamp to hand to once the wait returns. + public static long WaitStart() => Stopwatch.GetTimestamp(); + + /// Counts one wait at that began at . + public static void NoteWait(WaitSite site, long startTimestamp) + { + int index = (int)site; + Interlocked.Increment(ref _waitCounts[index]); + Interlocked.Add(ref _waitTicks[index], Stopwatch.GetTimestamp() - startTimestamp); + } + + public static long WaitCount(WaitSite site) => Interlocked.Read(ref _waitCounts[(int)site]); + + public static double WaitMilliseconds(WaitSite site) => + Interlocked.Read(ref _waitTicks[(int)site]) * 1000.0 / Stopwatch.Frequency; + + /// vkDeviceWaitIdle, counted at . Every call site goes through here. + public static Result WaitDeviceIdle(Vk api, Device device) + { + long start = Stopwatch.GetTimestamp(); + Result result = api.DeviceWaitIdle(device); + NoteWait(WaitSite.DeviceWaitIdle, start); + return result; + } + + /// One CPU frame interval (start of a frame to start of the next), in milliseconds. + public static void NoteFrameInterval(double milliseconds) => FrameIntervals.Add(milliseconds); + + /// + /// Takes and clears the counters, formatted as one sample (four lines joined + /// by '\n', no trailing newline), or null when the interval has not elapsed. + /// Called once per frame by the render thread. + /// + public static string? SampleIfDue(TimeSpan interval) + { + long now = Stopwatch.GetTimestamp(); + long last = Interlocked.Read(ref _lastSample); + if (last == 0) + { + Interlocked.CompareExchange(ref _lastSample, now, 0); + return null; + } + + double elapsed = (now - last) / (double)Stopwatch.Frequency; + if (elapsed < interval.TotalSeconds) return null; + if (Interlocked.CompareExchange(ref _lastSample, now, last) != last) return null; + + long frames = Interlocked.Exchange(ref _frames, 0); + long allocations = Interlocked.Exchange(ref _allocations, 0); + long uploads = Interlocked.Exchange(ref _uploads, 0); + long uploadTicks = Interlocked.Exchange(ref _uploadWaitTicks, 0); + long created = Interlocked.Exchange(ref _texturesCreated, 0); + long deleted = Interlocked.Exchange(ref _texturesDeleted, 0); + long dropped = Interlocked.Exchange(ref _droppedMeshWrites, 0); + long overflows = Interlocked.Exchange(ref _uniformOverflows, 0); + + var waitCounts = new long[WaitSiteCount]; + var waitMs = new double[WaitSiteCount]; + for (int i = 0; i < WaitSiteCount; i++) + { + waitCounts[i] = Interlocked.Exchange(ref _waitCounts[i], 0); + waitMs[i] = Interlocked.Exchange(ref _waitTicks[i], 0) * 1000.0 / Stopwatch.Frequency; + } + + var counters = new CounterSample( + BlockingUploads: Interlocked.Exchange(ref _blockingUploads, 0), + Uploads: Interlocked.Exchange(ref _uploadRequests, 0), + Scopes: Interlocked.Exchange(ref _scopesOpened, 0), + Barriers: Interlocked.Exchange(ref _imageBarriers, 0), + RebarFallbacks: Interlocked.Exchange(ref _rebarFallbacks, 0), + DynamicState: Interlocked.Exchange(ref _dynamicStateCommands, 0), + UniformRingUsed: Interlocked.Exchange(ref _uniformRingPeak, 0), + UniformRingCapacity: Interlocked.Read(ref _uniformRingCapacity), + BarrierCommands: Interlocked.Exchange(ref _barrierCommands, 0), + Frames: frames, + MaskRestarts: Interlocked.Exchange(ref _maskRestarts, 0), + FeedbackSplits: Interlocked.Exchange(ref _feedbackSplits, 0), + Passes: Interlocked.Exchange(ref _passes, 0), + PlanHits: Interlocked.Exchange(ref _planHits, 0), + PlanMisses: Interlocked.Exchange(ref _planMisses, 0), + InPassClears: Interlocked.Exchange(ref _inPassClears, 0), + PromotedClears: Interlocked.Exchange(ref _promotedClears, 0), + StandaloneClears: Interlocked.Exchange(ref _standaloneClears, 0), + PassSplits: Interlocked.Exchange(ref _passSplits, 0), + PushConstantWrites: Interlocked.Exchange(ref _intervalPushConstantWrites, 0), + StorageSetBinds: Interlocked.Exchange(ref _intervalStorageSetBinds, 0), + BindlessSlots: Interlocked.Exchange(ref _intervalBindlessSlots, 0), + BindlessPlaceholders: Interlocked.Exchange(ref _intervalBindlessPlaceholders, 0), + ComputePasses: Interlocked.Exchange(ref _computePasses, 0), + Dispatches: Interlocked.Exchange(ref _dispatches, 0), + NativePasses: Interlocked.Exchange(ref _intervalNativePasses, 0), + NativeDraws: Interlocked.Exchange(ref _intervalNativeDraws, 0), + NativeFullscreenDraws: Interlocked.Exchange(ref _intervalNativeFullscreenDraws, 0), + NativeMeshDraws: Interlocked.Exchange(ref _intervalNativeMeshDraws, 0), + NativeInstancedDraws: Interlocked.Exchange(ref _intervalNativeInstancedDraws, 0), + NativeIndirectDraws: Interlocked.Exchange(ref _intervalNativeIndirectDraws, 0)); + + double uploadMs = uploadTicks * 1000.0 / Stopwatch.Frequency; + + VulkanAllocator? memory = MemorySource; + MemorySnapshot memorySnapshot = memory == null ? default : memory.Snapshot(); + + return FormatIntervalLine(elapsed, frames, allocations, VulkanMemory.LiveAllocations, + uploads, uploadMs, created, deleted, dropped, overflows) + "\n" + + FormatPacingLine(FrameIntervals.Snapshot()) + "\n" + + FormatWaitsLine(waitCounts, waitMs) + "\n" + + FormatCountersLine(counters) + "\n" + + LatencyLine() + "\n" + + VulkanAllocator.FormatMemoryLine(memorySnapshot) + "\n" + + FormatTransientsLine(new TransientSample( + TransientBytes: (ulong)Interlocked.Read(ref _transientBytes), + AliasedBytes: (ulong)Interlocked.Exchange(ref _aliasedBytesPeak, 0), + HeapPeakBytes: memory?.TakeTransientHeapPeak() ?? 0, + Leases: Interlocked.Exchange(ref _transientLeases, 0), + AliasedLeases: Interlocked.Exchange(ref _aliasedLeases, 0), + ReadSelfCopies: Interlocked.Exchange(ref _readSelfCopies, 0), + ReadSelfPool: Interlocked.Read(ref _readSelfPool))) + "\n" + + FormatPipelinesLine(TakePipelineSample()) + "\n" + + FormatShadersLine(Interlocked.Read(ref _shadersNative), Interlocked.Read(ref _shadersRewritten), + Interlocked.Read(ref _shadersFailed)); + } + + private static long _shadersNative; + private static long _shadersRewritten; + private static long _shadersFailed; + + /// The counts of the latest shader load (docs/vulkan.md); the device reports them once per load. + public static void NoteShaderLoad(long native, long rewritten, long failed) + { + Interlocked.Exchange(ref _shadersNative, native); + Interlocked.Exchange(ref _shadersRewritten, rewritten); + Interlocked.Exchange(ref _shadersFailed, failed); + } + + /// + /// stats.shaders: the latest shader load's programs linked from the native manifest, through the + /// rewriter (no native program), and native programs that failed and fell back to the rewriter. + /// + public static string FormatShadersLine(long native, long rewritten, long failed) => + string.Format(CultureInfo.InvariantCulture, "stats.shaders native={0} rewritten={1} failed={2}", native, rewritten, failed); + + private static double Mib(ulong bytes) => bytes / (1024.0 * 1024.0); + + /// + /// stats.transients: transient image MiB at the last frame boundary, the + /// interval's largest aliased MiB in one frame, the Transient pool class's peak + /// block MiB, leases and aliased leases over the interval, ReadSelf copies taken + /// over the interval and the copies the pool holds. + /// + public static string FormatTransientsLine(TransientSample sample) => + string.Format(CultureInfo.InvariantCulture, + "stats.transients transient_mib={0:F1} aliased_mib={1:F1} heap_peak_mib={2:F1} leases={3} " + + "aliased_leases={4} readself_copies={5} readself_pool={6}", + Mib(sample.TransientBytes), Mib(sample.AliasedBytes), Mib(sample.HeapPeakBytes), sample.Leases, + sample.AliasedLeases, sample.ReadSelfCopies, sample.ReadSelfPool); + + /// + /// The allocator whose pool classes and heaps the stats.memory line + /// reports; the device sets it at init and clears it at dispose. + /// + public static volatile VulkanAllocator? MemorySource; + + /// The current device's bounded CPU timing reports. + public static volatile FrameTimingRecorder? LatencySource; + + /// + /// The five CPU intervals of , in the order they + /// appear on the stats.latency line. + /// + public static readonly string[] LatencyIntervalTokens = + { + "input", + "sim", + "render_submit", + "present", + "total", + }; + + public const int LatencyIntervalCount = 5; + + /// The intervals of one report, in order, in microseconds. + private static void IntervalsOf(in LatencyFrameReport report, ulong[] into) + { + into[0] = report.InputUs; + into[1] = report.SimulationUs; + into[2] = report.RenderSubmitUs; + into[3] = report.PresentUs; + into[4] = report.TotalUs; + } + + /// + /// Reduces the frame reports of one interval to a mean and a p99 per interval, + /// in milliseconds - the same reduction + /// applies to frame intervals, and the same nearest-rank percentile + /// (index ceil(0.99 * n) - 1), so the two lines are read the same way. + /// + public static void ReduceReports(LatencyFrameReport[] reports, double[] meanMs, double[] p99Ms) + { + for (int i = 0; i < LatencyIntervalCount; i++) + { + meanMs[i] = 0; + p99Ms[i] = 0; + } + if (reports.Length == 0) return; + + var values = new double[LatencyIntervalCount][]; + for (int i = 0; i < LatencyIntervalCount; i++) values[i] = new double[reports.Length]; + + var scratch = new ulong[LatencyIntervalCount]; + for (int r = 0; r < reports.Length; r++) + { + IntervalsOf(reports[r], scratch); + for (int i = 0; i < LatencyIntervalCount; i++) values[i][r] = scratch[i] / 1000.0; + } + + for (int i = 0; i < LatencyIntervalCount; i++) + { + double sum = 0; + for (int r = 0; r < reports.Length; r++) sum += values[i][r]; + meanMs[i] = sum / reports.Length; + + Array.Sort(values[i]); + int index = (int)Math.Ceiling(0.99 * reports.Length) - 1; + if (index < 0) index = 0; + if (index > reports.Length - 1) index = reports.Length - 1; + p99Ms[i] = values[i][index]; + } + } + + /// CPU phase durations reduced to a mean and p99, in milliseconds. + public static string FormatLatencyLine(int frames, double[] meanMs, double[] p99Ms) + { + var line = new StringBuilder("stats.latency frames="); + line.Append(frames.ToString(CultureInfo.InvariantCulture)); + for (int i = 0; i < LatencyIntervalCount; i++) + { + line.Append(' ').Append(LatencyIntervalTokens[i]).Append("_mean_ms=") + .Append(meanMs[i].ToString("F2", CultureInfo.InvariantCulture)); + line.Append(' ').Append(LatencyIntervalTokens[i]).Append("_p99_ms=") + .Append(p99Ms[i].ToString("F2", CultureInfo.InvariantCulture)); + } + return line.ToString(); + } + + /// The stats.latency line for the backend currently set as . + private static string LatencyLine() + { + FrameTimingRecorder? latency = LatencySource; + LatencyFrameReport[] reports = latency == null + ? Array.Empty() + : latency.TakeReports(); + + var meanMs = new double[LatencyIntervalCount]; + var p99Ms = new double[LatencyIntervalCount]; + ReduceReports(reports, meanMs, p99Ms); + + return FormatLatencyLine(reports.Length, meanMs, p99Ms); + } + + /// The original stats line. Its format must not change. + public static string FormatIntervalLine(double elapsed, long frames, long allocations, int liveAllocations, + long uploads, double uploadMs, long created, long deleted, long dropped, long overflows) + { + double frameMs = frames > 0 ? elapsed * 1000.0 / frames : 0; + + return string.Format( + System.Globalization.CultureInfo.InvariantCulture, + "stats {0:F1}s: {1} frames ({2:F1} ms/frame), {3} allocations ({4} live), " + + "{5} blocking uploads costing {6:F0} ms ({7:F0}% of the interval), " + + "textures +{8}/-{9}, mesh writes dropped {10}, uniform overflows {11}", + elapsed, frames, frameMs, allocations, liveAllocations, + uploads, uploadMs, uploadMs / (elapsed * 1000.0) * 100.0, created, deleted, dropped, overflows); + } + + public static string FormatPacingLine(FramePacingSnapshot pacing) => + string.Format(CultureInfo.InvariantCulture, + "stats.pacing samples={0} p50_ms={1:F3} p95_ms={2:F3} p99_ms={3:F3} stddev_ms={4:F3} stutters={5}", + pacing.Samples, pacing.P50, pacing.P95, pacing.P99, pacing.StdDev, pacing.Stutters); + + public static string FormatWaitsLine(long[] counts, double[] milliseconds) + { + var line = new StringBuilder("stats.waits"); + for (int i = 0; i < WaitSiteCount; i++) + { + line.Append(' ').Append(WaitSiteTokens[i]).Append("_n=") + .Append(counts[i].ToString(CultureInfo.InvariantCulture)); + line.Append(' ').Append(WaitSiteTokens[i]).Append("_ms=") + .Append(milliseconds[i].ToString("F1", CultureInfo.InvariantCulture)); + } + return line.ToString(); + } + + public static string FormatCountersLine(CounterSample counters) => + string.Format(CultureInfo.InvariantCulture, + "stats.counters blocking_uploads={0} uploads={1} scopes={2} barriers={3} rebar_fallbacks={4} " + + "dynamic_state={5} uniform_ring_used={6} uniform_ring_capacity={7} " + + "barrier_commands={8} barriers_per_frame={9:F1} mask_restarts={10} feedback_splits={11} " + + "passes={12} plan_hits={13} plan_misses={14} in_pass_clears={15} promoted_clears={16} " + + "standalone_clears={17} pass_splits={18} push_constants={19} storage_set_binds={20} " + + "bindless_slots={21} bindless_placeholders={22} compute_passes={23} dispatches={24}" + + " native_passes={25} native_draws={26} native_fullscreen_draws={27} " + + "native_mesh_draws={28} native_instanced_draws={29} native_indirect_draws={30}", + counters.BlockingUploads, counters.Uploads, counters.Scopes, counters.Barriers, + counters.RebarFallbacks, counters.DynamicState, counters.UniformRingUsed, counters.UniformRingCapacity, + counters.BarrierCommands, counters.Frames > 0 ? counters.Barriers / (double)counters.Frames : 0.0, + counters.MaskRestarts, counters.FeedbackSplits, counters.Passes, counters.PlanHits, counters.PlanMisses, + counters.InPassClears, counters.PromotedClears, counters.StandaloneClears, counters.PassSplits, + counters.PushConstantWrites, counters.StorageSetBinds, counters.BindlessSlots, counters.BindlessPlaceholders, + counters.ComputePasses, counters.Dispatches, counters.NativePasses, counters.NativeDraws, + counters.NativeFullscreenDraws, counters.NativeMeshDraws, counters.NativeInstancedDraws, + counters.NativeIndirectDraws); + + private static long _lastSample; + + private static long _pipelinesSync; + private static long _pipelinesAsync; + private static long _pipelinesPrewarmed; + private static long _pipelinesWarm; + private static long _pipelineDrawsSkipped; + private static long _pipelinesPending; + private static long _pipelineCacheBytes; + private static long _pipelineCacheSaves; + + /// A pipeline compiled on the calling (render) thread, blocking the draw. + public static void NotePipelineCompiledSync() => Interlocked.Increment(ref _pipelinesSync); + + /// A pipeline a draw asked for, compiled by the background worker. + public static void NotePipelineCompiledAsync() => Interlocked.Increment(ref _pipelinesAsync); + + /// A pipeline built from the pipeline-key log before any draw asked for it. + public static void NotePipelinePrewarmed() => Interlocked.Increment(ref _pipelinesPrewarmed); + + /// A FAIL_ON_PIPELINE_COMPILE_REQUIRED creation the driver cache satisfied without compiling. + public static void NotePipelineWarm() => Interlocked.Increment(ref _pipelinesWarm); + + /// A draw skipped because its pipeline was still compiling in the background. + public static void NotePipelineDrawSkipped() => Interlocked.Increment(ref _pipelineDrawsSkipped); + + public static void NotePipelinesPending(int pending) => Interlocked.Exchange(ref _pipelinesPending, pending); + + /// Serialised driver cache size at the last sample or save. + public static void NotePipelineCacheBytes(long bytes) => Interlocked.Exchange(ref _pipelineCacheBytes, bytes); + + /// A driver pipeline cache file written (opportunistic or shutdown). + public static void NotePipelineCacheSave() => Interlocked.Increment(ref _pipelineCacheSaves); + + public static long PipelineDrawsSkipped => Interlocked.Read(ref _pipelineDrawsSkipped); + + private static PipelineSample TakePipelineSample() => new( + CompiledSync: Interlocked.Exchange(ref _pipelinesSync, 0), + CompiledAsync: Interlocked.Exchange(ref _pipelinesAsync, 0), + Prewarmed: Interlocked.Exchange(ref _pipelinesPrewarmed, 0), + Warm: Interlocked.Exchange(ref _pipelinesWarm, 0), + DrawsSkipped: Interlocked.Exchange(ref _pipelineDrawsSkipped, 0), + Pending: Interlocked.Read(ref _pipelinesPending), + CacheBytes: Interlocked.Read(ref _pipelineCacheBytes), + Saves: Interlocked.Exchange(ref _pipelineCacheSaves, 0)); + + /// + /// stats.pipelines: pipelines compiled blocking, by the background worker, and + /// prewarmed from the key log over the interval; creations the driver cache satisfied + /// without a compile; draws skipped waiting for a pipeline; compiles still pending; the + /// serialised driver cache size at the last sample; cache files written over the interval. + /// + public static string FormatPipelinesLine(PipelineSample sample) => + string.Format(CultureInfo.InvariantCulture, + "stats.pipelines compiled_sync={0} compiled_async={1} prewarmed={2} warm={3} draws_skipped={4} " + + "pending={5} cache_bytes={6} saves={7}", + sample.CompiledSync, sample.CompiledAsync, sample.Prewarmed, sample.Warm, sample.DrawsSkipped, + sample.Pending, sample.CacheBytes, sample.Saves); +} + +/// The values on the stats.pipelines line. +internal readonly record struct PipelineSample( + long CompiledSync, + long CompiledAsync, + long Prewarmed, + long Warm, + long DrawsSkipped, + long Pending, + long CacheBytes, + long Saves); + +/// The per-interval counters on the stats.counters line. +internal readonly record struct CounterSample( + long BlockingUploads, + long Uploads, + long Scopes, + long Barriers, + long RebarFallbacks, + long DynamicState, + long UniformRingUsed, + long UniformRingCapacity, + long BarrierCommands = 0, + long Frames = 0, + long MaskRestarts = 0, + long FeedbackSplits = 0, + long Passes = 0, + long PlanHits = 0, + long PlanMisses = 0, + long InPassClears = 0, + long PromotedClears = 0, + long StandaloneClears = 0, + long PassSplits = 0, + long PushConstantWrites = 0, + long StorageSetBinds = 0, + long BindlessSlots = 0, + long BindlessPlaceholders = 0, + long ComputePasses = 0, + long Dispatches = 0, + long NativePasses = 0, + long NativeDraws = 0, + long NativeFullscreenDraws = 0, + long NativeMeshDraws = 0, + long NativeInstancedDraws = 0, + long NativeIndirectDraws = 0); + +/// The values on the stats.transients line. +internal readonly record struct TransientSample( + ulong TransientBytes, + ulong AliasedBytes, + ulong HeapPeakBytes, + long Leases, + long AliasedLeases, + long ReadSelfCopies, + long ReadSelfPool); + +/// Percentiles and spread of the frame-interval ring at one moment. +internal readonly record struct FramePacingSnapshot( + int Samples, double P50, double P95, double P99, double StdDev, int Stutters); + +/// +/// The last N CPU frame intervals, for p50/p95/p99, standard deviation and the +/// stutter count (intervals above 2 x p50). +/// +/// Both arrays are allocated once; adding a frame is a store under a lock, and a +/// snapshot sorts into the preallocated scratch array, so neither allocates. +/// Percentiles are nearest-rank (index ceil(p * n) - 1), the same rule the +/// client's OPTIMUM_FPS_LOG uses for p99, so the two logs are comparable. +/// +internal sealed class FrameIntervalRing +{ + public const int DefaultCapacity = 512; + + private readonly double[] _values; + private readonly double[] _scratch; + private readonly object _lock = new(); + private int _next; + private int _count; + + public FrameIntervalRing(int capacity) + { + if (capacity <= 0) throw new ArgumentOutOfRangeException(nameof(capacity)); + _values = new double[capacity]; + _scratch = new double[capacity]; + } + + public int Capacity => _values.Length; + + public int Count + { + get { lock (_lock) return _count; } + } + + public void Add(double milliseconds) + { + if (!(milliseconds >= 0) || double.IsInfinity(milliseconds)) return; + lock (_lock) + { + _values[_next] = milliseconds; + _next = (_next + 1) % _values.Length; + if (_count < _values.Length) _count++; + } + } + + public FramePacingSnapshot Snapshot() + { + lock (_lock) + { + int n = _count; + if (n == 0) return default; + + // The live values are the first n slots until the ring wraps, and all + // of them afterwards; order does not matter once they are sorted. + Array.Copy(_values, _scratch, n); + Array.Sort(_scratch, 0, n); + + double sum = 0; + for (int i = 0; i < n; i++) sum += _scratch[i]; + double mean = sum / n; + double squares = 0; + for (int i = 0; i < n; i++) + { + double delta = _scratch[i] - mean; + squares += delta * delta; + } + + double p50 = NearestRank(_scratch, n, 0.50); + int stutters = 0; + for (int i = n - 1; i >= 0 && _scratch[i] > 2.0 * p50; i--) stutters++; + + return new FramePacingSnapshot(n, p50, NearestRank(_scratch, n, 0.95), + NearestRank(_scratch, n, 0.99), Math.Sqrt(squares / n), stutters); + } + } + + private static double NearestRank(double[] sorted, int n, double percentile) + { + int index = (int)Math.Ceiling(percentile * n) - 1; + if (index < 0) index = 0; + if (index > n - 1) index = n - 1; + return sorted[index]; + } +} diff --git a/Optimum.Render.Vulkan/Core/WindowSurface.cs b/Optimum.Render.Vulkan/Core/WindowSurface.cs new file mode 100644 index 00000000..dbf5c8e9 --- /dev/null +++ b/Optimum.Render.Vulkan/Core/WindowSurface.cs @@ -0,0 +1,100 @@ +using System; +using System.Collections.Generic; +using OpenTK.Windowing.GraphicsLibraryFramework; +using Silk.NET.Vulkan; +using Silk.NET.Vulkan.Extensions.KHR; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// Creates a Vulkan presentation surface for a GLFW window. +/// +/// The client already opens its window through OpenTK's GLFW bindings, so the +/// surface is created from the same window pointer rather than by standing up a +/// second windowing stack. The only change the client needs is to ask for +/// ContextAPI.NoAPI so GLFW does not create an OpenGL context alongside. +/// +internal static unsafe class WindowSurface +{ + /// Whether a Vulkan loader is reachable at all. + public static bool VulkanSupported() + { + try + { + return GLFW.VulkanSupported(); + } + catch + { + return false; + } + } + + /// + /// The instance extensions the platform needs before a surface can be made. + /// These must be enabled at instance creation, which is why they are asked + /// for before the context exists. + /// + public static string[] RequiredInstanceExtensions() + { + try + { + return GLFW.GetRequiredInstanceExtensions() ?? Array.Empty(); + } + catch + { + return Array.Empty(); + } + } + + /// + /// Creates the surface. The window pointer is the GLFW handle the client + /// already holds. + /// + /// + /// Destroys a surface that never reached a . The + /// swapchain owns the surface once it exists, so this is only for the + /// failure paths between creation and hand-over; the instance must not be + /// destroyed with a surface still alive under it. + /// + public static void Destroy(VulkanContext context, SurfaceKHR surface) + { + if (surface.Handle == 0) return; + if (!context.Api.TryGetInstanceExtension(context.Instance, out KhrSurface surfaceApi)) return; + surfaceApi.DestroySurface(context.Instance, surface, null); + surfaceApi.Dispose(); + } + + public static bool TryCreate( + VulkanContext context, IntPtr windowHandle, out SurfaceKHR surface, out string? failureReason) + { + surface = default; + failureReason = null; + + if (windowHandle == IntPtr.Zero) + { + failureReason = "no window handle"; + return false; + } + + try + { + var instanceHandle = new VkHandle(context.Instance.Handle); + int result = GLFW.CreateWindowSurface( + instanceHandle, (Window*)windowHandle, null, out VkHandle surfaceHandle); + + if (result != 0) + { + failureReason = "glfwCreateWindowSurface failed with " + result; + return false; + } + + surface = new SurfaceKHR((ulong)surfaceHandle.Handle); + return true; + } + catch (Exception error) + { + failureReason = "surface creation threw: " + error.Message; + return false; + } + } +} diff --git a/Optimum.Render.Vulkan/Frame/FrameTimeline.cs b/Optimum.Render.Vulkan/Frame/FrameTimeline.cs new file mode 100644 index 00000000..7f539a25 --- /dev/null +++ b/Optimum.Render.Vulkan/Frame/FrameTimeline.cs @@ -0,0 +1,389 @@ +using System; +using System.Collections.Generic; +using System.Threading; +using Silk.NET.Vulkan; + +using Semaphore = Silk.NET.Vulkan.Semaphore; + +// The Frame/ folder follows the plan's layout; the namespace stays Core until the +// renderer is reorganised, so every existing consumer keeps its one using. +namespace Optimum.Render.Vulkan.Core; + +/// +/// The four numbers a needs from the timelines. An +/// interface so the lifetime rules can be tested without a device. +/// +internal interface ITimelineClock +{ + /// + /// The newest Frame value any command recorded or submitted so far carries: + /// a resource released now can only be referenced by work at or below it. + /// + ulong FrameRecorded { get; } + + /// The same for the Transfer timeline. + ulong TransferRecorded { get; } + + /// The Frame value the GPU has finished (the semaphore's counter). + ulong FrameCompleted { get; } + + /// The Transfer value the GPU has finished. + ulong TransferCompleted { get; } +} + +/// +/// The renderer's clock: two timeline semaphores. +/// +/// Every graphics submission of frame n signals to +/// n; transfer work signals . Pacing waits on the +/// Frame timeline, and deferred destruction compares recorded values against the +/// counters, so nothing in steady state needs a fence. +/// +/// Values are reserved before recording () and noted as +/// signalled once the submit that carries them was accepted +/// (). Waits are clamped to the signalled value: +/// waiting for a value no submission will ever signal would hang forever. +/// +/// Presentation cannot wait on a timeline, so the per-image present semaphores +/// stay binary and live in the swapchain. +/// +internal sealed unsafe class FrameTimeline : ITimelineClock, IDisposable +{ + private readonly VulkanContext _context; + private long _frameReserved; + private long _frameSignalled; + private long _transferReserved; + private long _transferSignalled; + private bool _disposed; + + public Semaphore Frame { get; } + public Semaphore Transfer { get; } + + public FrameTimeline(VulkanContext context) + { + _context = context; + Frame = CreateTimeline(context, "the Frame timeline semaphore"); + Transfer = CreateTimeline(context, "the Transfer timeline semaphore"); + } + + private static Semaphore CreateTimeline(VulkanContext context, string what) + { + var type = new SemaphoreTypeCreateInfo + { + SType = StructureType.SemaphoreTypeCreateInfo, + SemaphoreType = SemaphoreType.Timeline, + InitialValue = 0, + }; + var info = new SemaphoreCreateInfo + { + SType = StructureType.SemaphoreCreateInfo, + PNext = &type, + }; + Semaphore semaphore; + VulkanResult.Check(context.Api.CreateSemaphore(context.Device, &info, null, &semaphore), + "vkCreateSemaphore for " + what); + return semaphore; + } + + // ------------------------------------------------------------------ Frame + + /// The value the next frame's submission will signal. Render thread. + public ulong ReserveFrame() => (ulong)Interlocked.Increment(ref _frameReserved); + + /// The submission carrying was accepted by the queue. + public void NoteFrameSubmitted(ulong value) => RaiseTo(ref _frameSignalled, value); + + public ulong FrameRecorded => (ulong)Interlocked.Read(ref _frameReserved); + + /// The newest Frame value handed to an accepted submission. + public ulong FrameSignalled => (ulong)Interlocked.Read(ref _frameSignalled); + + public ulong FrameCompleted => Counter(Frame, "the Frame timeline"); + + /// + /// Blocks until the GPU finished Frame value (clamped + /// to what was signalled) and counts one wait at , even + /// when the value has already passed: the count is the number of pacing points, + /// which is what the stats gate compares. + /// + public void WaitForFrame(ulong value, WaitSite site) => + Wait(Frame, WaitTarget(value, FrameSignalled), site, "the Frame timeline"); + + /// + /// Teardown only: waits for every signalled frame before the caller destroys + /// what those frames name, and never throws (a lost device or an exception + /// unwinding past the caller's own idle wait must not turn into a driver crash + /// on destroying objects a queued command buffer still uses). + /// + public void WaitForSignalledFramesAtTeardown() => WaitAtTeardown(Frame, FrameSignalled); + + /// The Transfer timeline's counterpart of . + public void WaitForSignalledTransfersAtTeardown() => WaitAtTeardown(Transfer, TransferSignalled); + + private void WaitAtTeardown(Semaphore semaphore, ulong signalled) + { + Semaphore handle = semaphore; + ulong target = signalled; + var info = new SemaphoreWaitInfo + { + SType = StructureType.SemaphoreWaitInfo, + SemaphoreCount = 1, + PSemaphores = &handle, + PValues = &target, + }; + long waitStart = VulkanStats.WaitStart(); + _context.Api.WaitSemaphores(_context.Device, &info, 5UL * 1000 * 1000 * 1000); + VulkanStats.NoteWait(WaitSite.DeviceWaitIdle, waitStart); + } + + // --------------------------------------------------------------- Transfer + + /// The value the next transfer submission will signal. + public ulong ReserveTransfer() => (ulong)Interlocked.Increment(ref _transferReserved); + + public void NoteTransferSubmitted(ulong value) => RaiseTo(ref _transferSignalled, value); + + public ulong TransferRecorded => (ulong)Interlocked.Read(ref _transferReserved); + + public ulong TransferSignalled => (ulong)Interlocked.Read(ref _transferSignalled); + + public ulong TransferCompleted => Counter(Transfer, "the Transfer timeline"); + + public void WaitForTransfer(ulong value, WaitSite site) => + Wait(Transfer, WaitTarget(value, TransferSignalled), site, "the Transfer timeline"); + + // ------------------------------------------------------------------ rules + + /// Never wait past the newest signalled value; nothing would ever wake the wait. + public static ulong WaitTarget(ulong requested, ulong signalled) => Math.Min(requested, signalled); + + // ---------------------------------------------------------------- helpers + + private ulong Counter(Semaphore semaphore, string what) + { + ulong value; + VulkanResult.Check(_context.Api.GetSemaphoreCounterValue(_context.Device, semaphore, &value), + "vkGetSemaphoreCounterValue on " + what); + return value; + } + + private void Wait(Semaphore semaphore, ulong value, WaitSite site, string what) + { + Semaphore handle = semaphore; + ulong target = value; + var info = new SemaphoreWaitInfo + { + SType = StructureType.SemaphoreWaitInfo, + SemaphoreCount = 1, + PSemaphores = &handle, + PValues = &target, + }; + + long waitStart = VulkanStats.WaitStart(); + Result result = _context.Api.WaitSemaphores(_context.Device, &info, ulong.MaxValue); + VulkanStats.NoteWait(site, waitStart); + VulkanResult.Check(result, "vkWaitSemaphores on " + what); + } + + private static void RaiseTo(ref long field, ulong value) + { + long wanted = (long)Math.Min(value, long.MaxValue); + long seen = Interlocked.Read(ref field); + while (wanted > seen) + { + long previous = Interlocked.CompareExchange(ref field, wanted, seen); + if (previous == seen) break; + seen = previous; + } + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + _context.Api.DestroySemaphore(_context.Device, Frame, null); + _context.Api.DestroySemaphore(_context.Device, Transfer, null); + } +} + +/// +/// Resources waiting for the GPU to stop referencing them. +/// +/// Each entry records the newest Frame and Transfer values that existed when it +/// was retired: any command that could name the resource carries one of those +/// values or an older one. The entry is destroyed at the first +/// that sees both counters at or past its values, never +/// earlier, and ready entries are destroyed in the order they were retired. +/// +/// is safe from any thread (the game's VAO and UBO +/// finalizers release from the finalizer thread). runs on +/// the render thread at the start of a frame, and disposes outside the lock so a +/// resource whose Dispose retires something else cannot deadlock. +/// +internal sealed class RetireQueue +{ + private readonly record struct Entry(IDisposable Resource, ulong Frame, ulong Transfer); + + private readonly ITimelineClock _clock; + private readonly object _lock = new(); + private readonly List _entries = new(); + private int _count; + + public RetireQueue(ITimelineClock clock) => _clock = clock; + + /// Entries not yet destroyed. + public int PendingCount => Volatile.Read(ref _count); + + /// Queues against the timeline values recorded right now. + public void Retire(IDisposable resource) + { + lock (_lock) + { + _entries.Add(new Entry(resource, _clock.FrameRecorded, _clock.TransferRecorded)); + Volatile.Write(ref _count, _entries.Count); + } + } + + /// + /// Destroys every entry whose Frame and Transfer values have both completed, + /// oldest first. An entry that has not passed stays queued without holding back + /// later entries that have. Returns how many were destroyed. + /// + public int Collect() + { + if (PendingCount == 0) return 0; + + // Completion only moves forward, so counters read before taking the lock + // are still true for every entry inside it. + ulong frameCompleted = _clock.FrameCompleted; + ulong transferCompleted = _clock.TransferCompleted; + + List? ready = null; + lock (_lock) + { + int kept = 0; + for (int i = 0; i < _entries.Count; i++) + { + Entry entry = _entries[i]; + if (entry.Frame <= frameCompleted && entry.Transfer <= transferCompleted) + { + ready ??= new List(); + ready.Add(entry.Resource); + } + else + { + _entries[kept++] = entry; + } + } + _entries.RemoveRange(kept, _entries.Count - kept); + Volatile.Write(ref _count, _entries.Count); + } + + if (ready == null) return 0; + foreach (IDisposable resource in ready) resource.Dispose(); + return ready.Count; + } + + /// + /// Destroys everything regardless of the timelines, oldest first. Teardown + /// only, after the GPU has finished all submitted work. + /// + public void DisposeAll() + { + while (true) + { + Entry[] all; + lock (_lock) + { + if (_entries.Count == 0) return; + all = _entries.ToArray(); + _entries.Clear(); + Volatile.Write(ref _count, 0); + } + foreach (Entry entry in all) entry.Resource.Dispose(); + } + } +} + +/// +/// Tells a resource created in the last few frames from a long-lived one, with +/// no per-resource storage. +/// +/// Resource ids () only ever increase, so the highest id +/// issued when a frame began is a watermark: every id above the watermark of the +/// frame N - 1 frames back was created within the last N frames. The class +/// keeps one watermark per frame in a ring. +/// +/// The descriptor layer uses it to route sets naming short-lived resources (GUI +/// text, atlas tasks, fresh chunk meshes, overflow uniform copies) to the per-slot +/// arena that is reset every frame, instead of caching them in +/// only to evict them moments later. +/// +internal sealed class ResourceAge +{ + public const int DefaultShortLivedFrames = 60; + + private readonly ulong[] _watermarks; + private long _frames; + private int _shortLivedFrames; + + public ResourceAge(int capacity = DefaultShortLivedFrames) + { + if (capacity <= 0) throw new ArgumentOutOfRangeException(nameof(capacity)); + _watermarks = new ulong[capacity]; + _shortLivedFrames = capacity; + } + + /// + /// How many frames a resource counts as short-lived for, at most the ring's + /// capacity. Zero makes every resource long-lived (the arena is never used). + /// + public int ShortLivedFrames + { + get => _shortLivedFrames; + set + { + if (value < 0 || value > _watermarks.Length) throw new ArgumentOutOfRangeException(nameof(value)); + _shortLivedFrames = value; + } + } + + /// Frames noted so far. + public long Frames => _frames; + + /// Records the watermark at the start of a frame: the highest resource id issued so far. + public void NoteFrame(ulong highestIssuedId) + { + _watermarks[_frames % _watermarks.Length] = highestIssuedId; + _frames++; + } + + /// + /// Whether was created within the last + /// frames, the current one included. Id 0 (a + /// permanent resource) never is. Before that many frames have been noted, + /// every resource is: none can be older. + /// + public bool IsShortLived(ulong resource) + { + if (resource == 0 || _shortLivedFrames == 0) return false; + if (_frames < _shortLivedFrames) return true; + + ulong watermark = _watermarks[(_frames - _shortLivedFrames) % _watermarks.Length]; + return resource > watermark; + } + + /// Whether any resource a set names is short-lived. + public bool NamesShortLived(DescriptorSetContents contents) + { + foreach (SamplerBindingValue sampler in contents.Samplers) + { + if (IsShortLived(sampler.Resource)) return true; + } + foreach (BufferBindingValue buffer in contents.Buffers) + { + if (IsShortLived(buffer.Resource)) return true; + } + return false; + } +} diff --git a/Optimum.Render.Vulkan/Frame/IndirectRing.cs b/Optimum.Render.Vulkan/Frame/IndirectRing.cs new file mode 100644 index 00000000..0c5bf18b --- /dev/null +++ b/Optimum.Render.Vulkan/Frame/IndirectRing.cs @@ -0,0 +1,141 @@ +using System; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// The bookkeeping of the per-slot indirect-command ring, with no Vulkan in it. +/// +/// Each frame slot owns one indirect buffer. Multi-draws of a frame bump-allocate +/// from the slot's buffer, and the cursor returns to zero only when the slot +/// begins its next frame, which is after the Frame timeline says the GPU has +/// finished the previous frame that used it. The ring never wraps inside a frame: +/// the wrapping ring this replaces could hand a new draw the region a frame still +/// in flight was reading. +/// +/// A frame that asks for more than its slot holds is told so ( +/// returns false) and the caller takes an overflow buffer for the rest of that +/// frame. The slot's buffer grows only at a frame boundary, to fit the busiest +/// frame any slot has seen. +/// +internal sealed class IndirectRing +{ + /// Smallest buffer a slot is given: 13107 indexed indirect commands. + public const ulong DefaultMinimumCapacity = 256UL * 1024; + + private const ulong Granularity = 64UL * 1024; + + private struct Slot + { + public ulong Capacity; + public ulong Cursor; + public ulong Usage; + public int Overflows; + } + + private readonly Slot[] _slots; + private int _current = -1; + + public IndirectRing(int slots, ulong minimumCapacity = DefaultMinimumCapacity) + { + if (slots <= 0) throw new ArgumentOutOfRangeException(nameof(slots)); + _slots = new Slot[slots]; + MinimumCapacity = Math.Max(1UL, minimumCapacity); + } + + public ulong MinimumCapacity { get; } + + /// The slot recording now, -1 before the first frame. + public int Current => _current; + + /// Bytes the busiest frame so far asked for, overflow included. Never shrinks. + public ulong PeakFrameUsage { get; private set; } + + public ulong CapacityOf(int slot) => _slots[slot].Capacity; + public ulong CursorOf(int slot) => _slots[slot].Cursor; + public ulong FrameUsageOf(int slot) => _slots[slot].Usage; + + /// Allocations of the slot's current frame that did not fit its buffer. + public int OverflowsOf(int slot) => _slots[slot].Overflows; + + /// + /// A buffer size that holds bytes with half again as + /// much headroom, rounded to 64 KiB, and never below the minimum. + /// + public ulong CapacityFor(ulong demand) + { + // A ring whose minimum is below the granularity (tests) rounds to its minimum. + ulong granularity = Math.Min(Granularity, MinimumCapacity); + ulong wanted = demand + demand / 2; + ulong rounded = (wanted + granularity - 1) / granularity * granularity; + return Math.Max(MinimumCapacity, rounded); + } + + /// + /// Starts 's next frame: folds every slot's last frame + /// usage into the peak, resets this slot's cursor, and reports whether its + /// existing buffer is too small for the peak. When it is, the caller retires + /// the old buffer, creates one of bytes and calls + /// . A slot without a buffer is not grown here; it gets + /// one at its first allocation. + /// + public bool BeginFrame(int slot, out ulong capacity) + { + foreach (Slot previous in _slots) PeakFrameUsage = Math.Max(PeakFrameUsage, previous.Usage); + + _current = slot; + ref Slot current = ref _slots[slot]; + current.Cursor = 0; + current.Usage = 0; + current.Overflows = 0; + + capacity = CapacityFor(PeakFrameUsage); + return current.Capacity != 0 && current.Capacity < PeakFrameUsage; + } + + /// + /// Whether the current slot has no buffer yet, and the size to create when it + /// has none. Creating one is safe at any time: nothing recorded names it. + /// + public bool NeedsBuffer(ulong bytes, out ulong capacity) + { + RequireFrame(); + capacity = CapacityFor(Math.Max(PeakFrameUsage, _slots[_current].Usage + bytes)); + return _slots[_current].Capacity == 0; + } + + /// Records that the current slot's buffer now holds bytes. + public void Attach(ulong capacity) + { + RequireFrame(); + if (capacity < _slots[_current].Cursor) throw new InvalidOperationException("a slot buffer cannot shrink under its cursor"); + _slots[_current].Capacity = capacity; + } + + /// + /// Bump-allocates in the current slot's buffer. Never + /// wraps: when the rest of the buffer is too small, returns false and leaves the + /// cursor where it is. Either way the bytes count toward this frame's usage. + /// + public bool TryAllocate(ulong bytes, out ulong offset) + { + RequireFrame(); + ref Slot current = ref _slots[_current]; + current.Usage += bytes; + + if (current.Cursor + bytes > current.Capacity) + { + current.Overflows++; + offset = 0; + return false; + } + + offset = current.Cursor; + current.Cursor += bytes; + return true; + } + + private void RequireFrame() + { + if (_current < 0) throw new InvalidOperationException("BeginFrame has not been called yet"); + } +} diff --git a/Optimum.Render.Vulkan/Frame/QueryRing.cs b/Optimum.Render.Vulkan/Frame/QueryRing.cs new file mode 100644 index 00000000..5089ac61 --- /dev/null +++ b/Optimum.Render.Vulkan/Frame/QueryRing.cs @@ -0,0 +1,421 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +// The Frame/ folder follows the plan's layout; the namespace stays Core until the +// renderer is reorganised. +namespace Optimum.Render.Vulkan.Core; + +/// +/// Occlusion queries that never make the CPU wait. +/// +/// Each frame slot owns its own occlusion query pools, 32 queries apiece, reset +/// wholesale when the slot starts a frame (the first commands of its command +/// buffer, outside any rendering scope). A GL query object is a small record +/// pointing at the slot, indices and Frame timeline value of its most recently +/// ended query. Results are read with vkGetQueryPoolResults without the +/// wait bit, with availability, into the slot's host buffer, and only once the +/// timeline says the command buffer carrying the query has finished: either when +/// the game polls after that point, or at the latest when the slot is recycled, +/// just before its reset. The result therefore appears a frame or two after the +/// query, like GL's availability polling, and nothing ever blocks on it. +/// +/// A GL query counts every sample between begin and end, across framebuffer +/// binds; a Vulkan query has to begin and end inside one rendering scope and one +/// command buffer. So a running query is suspended when its scope closes (a +/// target change, a layout transition, a readback or upload that submits the +/// frame partially, present) and resumed on a fresh index when the next scope +/// opens. Its result is the sum of those segments. The pool a resumed segment +/// needs is created and reset between the two scopes. +/// +/// The plan named vkCmdCopyQueryPoolResults for the copy. That command is +/// not part of Vulkan 1.3 core and needs an extension the hardware floor does not +/// include; a host read gated on the timeline gives the same no-wait guarantee. +/// +internal sealed unsafe class QueryRing : IDisposable +{ + public const uint QueriesPerPool = 32; + + private const QueryResultFlags ReadFlags = QueryResultFlags.Result64Bit | QueryResultFlags.ResultWithAvailabilityBit; + private const int Stride = 2 * sizeof(ulong); + + private readonly VulkanContext _context; + private readonly ITimelineClock _clock; + private readonly SlotQueries[] _slots; + private readonly Dictionary _objects = new(); + private int _nextId = 1; + private int _currentSlot = -1; + // At most one occlusion query is active at a time, as in GL: either running + // inside the open scope, or suspended until the next scope opens. + private QueryRecord? _running; + private QueryRecord? _suspended; + private bool _disposed; + + public QueryRing(VulkanContext context, ITimelineClock clock, int framesInFlight) + { + _context = context; + _clock = clock; + _slots = new SlotQueries[framesInFlight]; + for (int i = 0; i < framesInFlight; i++) _slots[i] = new SlotQueries(); + } + + private sealed class SlotQueries + { + public readonly List Pools = new(); + /// Query indices handed out in the current generation. + public uint Used; + /// Bumped at every frame start of the slot; a record of an older generation was harvested. + public ulong Generation; + /// Records of the current generation not yet harvested. + public readonly List Pending = new(); + /// Value and availability per query, two ulongs each, in pool order. + public ulong[] Host = Array.Empty(); + } + + private sealed class QueryRecord + { + public readonly int Slot; + public readonly ulong Generation; + /// One index per segment: a new one each time the query resumed in a new scope. + public readonly List Indices = new(1); + public bool Ended; + /// A segment could not be resumed; the result reports every sample passed. + public bool Lost; + /// The query object was deleted while the query was running. + public bool Abandoned; + public ulong FrameValue; + public bool Resolved; + public ulong Samples; + + public QueryRecord(int slot, ulong generation) + { + Slot = slot; + Generation = generation; + } + } + + private sealed class QueryObject + { + public QueryRecord? Active; + public QueryRecord? Latest; + /// The result of an earlier query, returned while the latest one is still in flight. + public int PreviousResult = int.MaxValue; + } + + /// Pools created so far across every slot. Tests only. + internal int PoolCount + { + get + { + int count = 0; + foreach (SlotQueries slot in _slots) count += slot.Pools.Count; + return count; + } + } + + public int Create() + { + int id = _nextId++; + _objects[id] = new QueryObject(); + return id; + } + + /// + /// Forgets the query object. Its records stay with their slot until that slot + /// is recycled; the pools are shared, so there is nothing to destroy. A query + /// still running ends with its scope and is not resumed. + /// + public void Delete(int id) + { + if (!_objects.Remove(id, out QueryObject? query) || query.Active == null) return; + if (ReferenceEquals(query.Active, _suspended)) _suspended = null; + else if (ReferenceEquals(query.Active, _running)) query.Active.Abandoned = true; + } + + /// + /// The slot is starting a frame and the timeline has passed its previous one: + /// reads every result that frame produced into the host buffer, then records + /// the pool resets at the top of the new command buffer. + /// + public void BeginSlot(int slotIndex, CommandBuffer commandBuffer) + { + SlotQueries slot = _slots[slotIndex]; + Harvest(slot); + + uint poolsUsed = (slot.Used + QueriesPerPool - 1) / QueriesPerPool; + for (int i = 0; i < poolsUsed; i++) + { + _context.Api.CmdResetQueryPool(commandBuffer, slot.Pools[i], 0, QueriesPerPool); + } + + slot.Used = 0; + slot.Generation++; + _currentSlot = slotIndex; + // Present closed the last scope, so nothing is running; a query the + // previous frame never ended is reported lost by its own slot's harvest. + _running = null; + _suspended = null; + } + + /// Whether can begin now: known, in a frame, and no other query active. + public bool CanBegin(int id) => + _currentSlot >= 0 && _running == null && _suspended == null && _objects.ContainsKey(id); + + /// The next query index lives in a pool that does not exist yet. + public bool NextNeedsPool => _currentSlot >= 0 && NeedsPool(_slots[_currentSlot]); + + private static bool NeedsPool(SlotQueries slot) => slot.Used / QueriesPerPool >= (uint)slot.Pools.Count; + + /// + /// Creates the pool the next index needs and records its reset. The caller is + /// outside any rendering scope. Every later frame of the slot resets it with + /// the others. + /// + public void AddPool(CommandBuffer commandBuffer) + { + SlotQueries slot = _slots[_currentSlot]; + var createInfo = new QueryPoolCreateInfo + { + SType = StructureType.QueryPoolCreateInfo, + QueryType = QueryType.Occlusion, + QueryCount = QueriesPerPool, + }; + QueryPool created; + VulkanResult.Check(_context.Api.CreateQueryPool(_context.Device, &createInfo, null, &created), + "vkCreateQueryPool for the occlusion query ring"); + slot.Pools.Add(created); + + var host = new ulong[slot.Pools.Count * QueriesPerPool * 2]; + Array.Copy(slot.Host, host, slot.Host.Length); + slot.Host = host; + + _context.Api.CmdResetQueryPool(commandBuffer, created, 0, QueriesPerPool); + } + + /// + /// Begins a query for . Inside an open scope it starts + /// at once; outside one it starts when the next scope opens. The caller + /// checked and added a pool if . + /// + public void Begin(int id, CommandBuffer commandBuffer, bool scopeOpen) + { + if (!CanBegin(id) || NextNeedsPool) return; + + SlotQueries slot = _slots[_currentSlot]; + var record = new QueryRecord(_currentSlot, slot.Generation); + slot.Pending.Add(record); + _objects[id].Active = record; + + if (scopeOpen) Start(record, commandBuffer); + else _suspended = record; + } + + /// + /// Ends the query begun for in this frame, whose last + /// segment is recorded in the command buffer that will signal + /// . It becomes the query whose result the object reports. + /// + public void End(int id, ulong frameValue, CommandBuffer commandBuffer) + { + if (!_objects.TryGetValue(id, out QueryObject? query) || query.Active == null) return; + + QueryRecord record = query.Active; + query.Active = null; + if (ReferenceEquals(record, _running)) + { + _running = null; + Stop(record, commandBuffer); + } + else if (ReferenceEquals(record, _suspended)) + { + _suspended = null; + } + + // A query begun in an earlier frame cannot be ended in this one; Vulkan + // requires both in the same command buffer. Harvest reports it as lost. + if (record.Slot != _currentSlot || record.Generation != _slots[record.Slot].Generation) return; + + if (query.Latest != null) + { + Refresh(query.Latest); + if (query.Latest.Resolved) query.PreviousResult = Clamp(query.Latest.Samples); + } + + record.Ended = true; + record.FrameValue = frameValue; + query.Latest = record; + } + + /// Scope hook, before vkCmdEndRendering: a running query ends its segment and waits for the next scope. + public void OnScopeClosing(CommandBuffer commandBuffer) + { + QueryRecord? record = _running; + if (record == null) return; + _running = null; + Stop(record, commandBuffer); + if (!record.Abandoned) _suspended = record; + } + + /// Scope hook, after vkCmdEndRendering: the pool a resumed segment will need is reset now, outside the scope. + public void OnScopeClosed(CommandBuffer commandBuffer) + { + if (_suspended != null && NextNeedsPool) AddPool(commandBuffer); + } + + /// Scope hook, after vkCmdBeginRendering: a suspended query resumes on a fresh index. + public void OnScopeOpened(CommandBuffer commandBuffer) + { + QueryRecord? record = _suspended; + if (record == null) return; + _suspended = null; + if (record.Slot != _currentSlot || record.Generation != _slots[record.Slot].Generation) return; + + // Unreachable when every scope end ran OnScopeClosed; a query that cannot + // resume reports every sample passed rather than a partial count. + if (NextNeedsPool) + { + record.Lost = true; + return; + } + Start(record, commandBuffer); + } + + private void Start(QueryRecord record, CommandBuffer commandBuffer) + { + SlotQueries slot = _slots[record.Slot]; + uint index = slot.Used++; + record.Indices.Add(index); + // GL_SAMPLES_PASSED is an exact count (sun glare divides it by 1500). + _context.Api.CmdBeginQuery(commandBuffer, slot.Pools[(int)(index / QueriesPerPool)], index % QueriesPerPool, + _context.Capabilities.OcclusionQueryPrecise ? QueryControlFlags.PreciseBit : default(QueryControlFlags)); + _running = record; + } + + private void Stop(QueryRecord record, CommandBuffer commandBuffer) + { + SlotQueries slot = _slots[record.Slot]; + uint index = record.Indices[record.Indices.Count - 1]; + _context.Api.CmdEndQuery(commandBuffer, slot.Pools[(int)(index / QueriesPerPool)], index % QueriesPerPool); + } + + /// GL_QUERY_RESULT_AVAILABLE: the latest ended query's command buffer has finished. + public bool IsResultAvailable(int id) + { + if (!_objects.TryGetValue(id, out QueryObject? query) || query.Latest == null) return false; + Refresh(query.Latest); + return query.Latest.Resolved; + } + + /// + /// The latest query's sample count once available; before that, the previous + /// query's, and "every sample passed" when there never was one - for a query + /// that gates culling or glare, visible is the failure that costs little. + /// + public int GetResult(int id) + { + if (!_objects.TryGetValue(id, out QueryObject? query)) return 0; + if (query.Latest != null) + { + Refresh(query.Latest); + if (query.Latest.Resolved) return Clamp(query.Latest.Samples); + } + return query.PreviousResult; + } + + private static int Clamp(ulong samples) => (int)Math.Min(samples, int.MaxValue); + + /// Reads one result early, once the timeline has passed the command buffer that carried it. + private void Refresh(QueryRecord record) + { + if (record.Resolved || !record.Ended) return; + SlotQueries slot = _slots[record.Slot]; + // Harvest resolves every record of a generation before bumping it. + if (record.Generation != slot.Generation) return; + if (_clock.FrameCompleted < record.FrameValue) return; + + if (record.Lost || record.Indices.Count == 0) + { + record.Resolved = true; + record.Samples = ulong.MaxValue; + return; + } + + ulong samples = 0; + fixed (ulong* host = slot.Host) + { + foreach (uint index in record.Indices) + { + uint base2 = index * 2; + Result status = _context.Api.GetQueryPoolResults(_context.Device, + slot.Pools[(int)(index / QueriesPerPool)], index % QueriesPerPool, 1, + (nuint)Stride, host + base2, (ulong)Stride, ReadFlags); + if (status != Result.Success && status != Result.NotReady) + { + VulkanResult.Check(status, "vkGetQueryPoolResults for an occlusion query"); + } + if (host[base2 + 1] == 0) return; + samples += host[base2]; + } + } + + record.Resolved = true; + record.Samples = samples; + } + + private void Harvest(SlotQueries slot) + { + if (slot.Pending.Count == 0) return; + + uint remaining = slot.Used; + fixed (ulong* host = slot.Host) + { + for (int p = 0; remaining > 0 && p < slot.Pools.Count; p++) + { + uint count = Math.Min(remaining, QueriesPerPool); + Result status = _context.Api.GetQueryPoolResults(_context.Device, slot.Pools[p], 0, count, + (nuint)(count * Stride), host + (ulong)p * QueriesPerPool * 2, (ulong)Stride, ReadFlags); + // NOT_READY only means some query was never ended; the availability + // word says which, and the rest are written. + if (status != Result.Success && status != Result.NotReady) + { + VulkanResult.Check(status, "vkGetQueryPoolResults for the occlusion query ring"); + } + remaining -= count; + } + } + + foreach (QueryRecord record in slot.Pending) + { + if (record.Resolved) continue; + record.Resolved = true; + record.Samples = SumOfHarvested(slot, record); + } + slot.Pending.Clear(); + } + + private static ulong SumOfHarvested(SlotQueries slot, QueryRecord record) + { + if (!record.Ended || record.Lost || record.Indices.Count == 0) return ulong.MaxValue; + ulong samples = 0; + foreach (uint index in record.Indices) + { + uint base2 = index * 2; + if (slot.Host[base2 + 1] == 0) return ulong.MaxValue; + samples += slot.Host[base2]; + } + return samples; + } + + /// The caller has waited for every signalled frame first. + public void Dispose() + { + if (_disposed) return; + _disposed = true; + foreach (SlotQueries slot in _slots) + { + foreach (QueryPool pool in slot.Pools) _context.Api.DestroyQueryPool(_context.Device, pool, null); + slot.Pools.Clear(); + } + _objects.Clear(); + } +} diff --git a/Optimum.Render.Vulkan/Graph/BarrierBatcher.cs b/Optimum.Render.Vulkan/Graph/BarrierBatcher.cs new file mode 100644 index 00000000..5ebf00f4 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/BarrierBatcher.cs @@ -0,0 +1,142 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Graph; + +/// +/// Collects the image barriers a group of uses needs and records them as one +/// vkCmdPipelineBarrier2. Stages and accesses come from each image's +/// , never ALL_COMMANDS. +/// +/// updates the tracker at once, so the barriers must be +/// flushed before the commands that perform the uses are recorded, and never +/// inside an open rendering scope (checked in debug builds). One batcher belongs +/// to one recording thread; the trackers it touches are locked per call. +/// +internal sealed unsafe class BarrierBatcher +{ + private readonly Vk _api; + private readonly List _scratch = new(); + private ImageMemoryBarrier2[] _pending = new ImageMemoryBarrier2[16]; + private int _count; + + public BarrierBatcher(Vk api) => _api = api; + + /// + /// Whether a rendering scope is open in the given command buffer. A flush + /// there is a transition inside a scope, which debug builds reject. + /// + public Func? ScopeOpen { get; set; } + + /// Barriers recorded by and not yet flushed. + public int Pending => _count; + + public void Require(VulkanTexture texture, uint baseMip, uint mipCount, uint baseLayer, uint layerCount, + ResourceUsage usage) => + Require(texture, baseMip, mipCount, baseLayer, layerCount, usage, discard: false); + + public void Require(VulkanTexture texture, uint baseMip, uint mipCount, uint baseLayer, uint layerCount, + ResourceUsage usage, bool discard) => + Require(texture.Image, texture.Aspect, texture.Sync, baseMip, mipCount, baseLayer, layerCount, usage, discard); + + /// An image the texture table does not own (a swapchain image), with its own tracker. + public void Require(Image image, ImageAspectFlags aspect, ResourceStateTracker tracker, + uint baseMip, uint mipCount, uint baseLayer, uint layerCount, ResourceUsage usage, bool discard) + { + lock (tracker) + { + _scratch.Clear(); + if (tracker.Require(baseMip, mipCount, baseLayer, layerCount, usage, discard, _scratch) == 0) return; + + foreach (ImageTransition transition in _scratch) + { + Append(image, aspect, transition); + } + } + } + + private void Append(Image image, ImageAspectFlags aspect, ImageTransition transition) + { + BarrierSides sides = transition.Sides; + var range = new ImageSubresourceRange(aspect, transition.BaseMip, transition.MipCount, + transition.BaseLayer, transition.LayerCount); + + // A second use of the same range before the flush (one texture attached + // to two slots in different roles): one barrier per subresource per + // call, so the two chain into one from the first source to the last + // destination. + for (int i = 0; i < _count; i++) + { + ref ImageMemoryBarrier2 existing = ref _pending[i]; + if (existing.Image.Handle != image.Handle || !SameRange(existing.SubresourceRange, range)) continue; + existing.NewLayout = sides.NewLayout; + existing.DstStageMask = sides.DstStage; + existing.DstAccessMask = sides.DstAccess; + return; + } + + if (_count == _pending.Length) Array.Resize(ref _pending, _pending.Length * 2); + _pending[_count++] = new ImageMemoryBarrier2 + { + SType = StructureType.ImageMemoryBarrier2, + SrcStageMask = sides.SrcStage, + SrcAccessMask = sides.SrcAccess, + DstStageMask = sides.DstStage, + DstAccessMask = sides.DstAccess, + OldLayout = sides.OldLayout, + NewLayout = sides.NewLayout, + SrcQueueFamilyIndex = Vk.QueueFamilyIgnored, + DstQueueFamilyIndex = Vk.QueueFamilyIgnored, + Image = image, + SubresourceRange = range, + }; + } + + private static bool SameRange(ImageSubresourceRange a, ImageSubresourceRange b) => + a.AspectMask == b.AspectMask && a.BaseMipLevel == b.BaseMipLevel && a.LevelCount == b.LevelCount && + a.BaseArrayLayer == b.BaseArrayLayer && a.LayerCount == b.LayerCount; + + /// Records every pending barrier as one vkCmdPipelineBarrier2; nothing when none is pending. + public void Flush(CommandBuffer commandBuffer) + { + if (_count == 0) return; + +#if DEBUG + if (ScopeOpen?.Invoke(commandBuffer) == true) + { + _count = 0; + throw new InvalidOperationException("image barriers flushed inside an open rendering scope"); + } +#endif + + fixed (ImageMemoryBarrier2* barriers = _pending) + { + var dependency = new DependencyInfo + { + SType = StructureType.DependencyInfo, + ImageMemoryBarrierCount = (uint)_count, + PImageMemoryBarriers = barriers, + }; + _api.CmdPipelineBarrier2(commandBuffer, &dependency); + } + + if (RenderTrace.Enabled) + { + for (int i = 0; i < _count; i++) + { + ImageMemoryBarrier2 b = _pending[i]; + RenderTrace.Write("barrier image=" + b.Image.Handle.ToString("x") + " mips=" + + b.SubresourceRange.BaseMipLevel + "+" + b.SubresourceRange.LevelCount + " layers=" + + b.SubresourceRange.BaseArrayLayer + "+" + b.SubresourceRange.LayerCount + " " + + b.OldLayout + "->" + b.NewLayout + " src=" + b.SrcStageMask + "/" + b.SrcAccessMask + + " dst=" + b.DstStageMask + "/" + b.DstAccessMask); + } + } + + VulkanStats.NoteImageBarriers(_count); + VulkanStats.NoteBarrierCommand(); + _count = 0; + } +} diff --git a/Optimum.Render.Vulkan/Graph/ComputePass.cs b/Optimum.Render.Vulkan/Graph/ComputePass.cs new file mode 100644 index 00000000..377d7ca1 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/ComputePass.cs @@ -0,0 +1,216 @@ +using System; +using System.Collections.Generic; + +namespace Optimum.Render.Vulkan.Graph; + +/// How a compute pass uses one bound image. +public enum ComputeAccess +{ + /// A combined image sampler over levels: SHADER_READ_ONLY_OPTIMAL. + Sampled, + /// A storage image read with imageLoad and never written: GENERAL. + StorageRead, + /// A storage image written with imageStore without reading it first: GENERAL. + StorageWrite, + /// A storage image read and written: GENERAL. + StorageReadWrite, +} + +/// +/// One image a compute pass binds: the descriptor binding in the program's pass set, +/// the texture, how it is used and which levels. A storage binding names exactly one +/// level (a storage image descriptor is one level); a sampled binding names a range, +/// and only that range moves to SHADER_READ_ONLY_OPTIMAL, so a prefilter can sample +/// level n and store level n + 1 of the same image in one pass. +/// +/// The layout(binding = N) of the pass set. +/// A texture id of the device. +/// What the dispatch does with it. +/// The first level the binding covers. +/// The number of levels; exactly 1 for storage. +/// Sampled only: linear min/mag filtering instead of nearest (always clamp to edge, nearest mip). +public readonly record struct ComputeBinding( + uint Binding, int TextureId, ComputeAccess Access, uint BaseMip = 0, uint MipCount = 1, bool Linear = false); + +/// +/// One vkCmdDispatch of a pass. Either explicit group counts, or +/// naming the index (into ) +/// of the image whose level extent the groups must cover with the program's local size. +/// +public sealed class ComputeDispatch +{ + public uint GroupsX = 1; + public uint GroupsY = 1; + public uint GroupsZ = 1; + + /// Index into the pass's bindings whose level extent sizes the dispatch; -1 uses the explicit counts. + public int SizeFromBinding = -1; + + /// Pushed for the compute stage before the dispatch; at most the program's push constant size. + public byte[]? PushConstants; + + public static ComputeDispatch Explicit(uint x, uint y, uint z = 1, byte[]? pushConstants = null) => + new() { GroupsX = x, GroupsY = y, GroupsZ = z, PushConstants = pushConstants }; + + public static ComputeDispatch Covering(int bindingIndex, byte[]? pushConstants = null) => + new() { SizeFromBinding = bindingIndex, PushConstants = pushConstants }; +} + +/// +/// A compute pass as its owner declares it (the frame graph's second pass kind; the +/// AO is its first user). The recorder closes any open rendering scope, queues one +/// barrier per binding from its (GENERAL for storage, +/// SHADER_READ_ONLY_OPTIMAL for sampled, per level), flushes them as one command, +/// binds the pipeline for and one descriptor set, and +/// records every dispatch. No rendering scope is open at any point of it. +/// +public sealed class ComputePassDeclaration +{ + public string Name = ""; + + /// A program from VulkanDevice.CreateComputeProgram. + public int ProgramId; + + /// Specialization constant values; constant_id = i takes element i (4 bytes each: uint, int, float bits or bool). + public uint[] Specialization = Array.Empty(); + + public ComputeBinding[] Bindings = Array.Empty(); + + public ComputeDispatch[] Dispatches = Array.Empty(); +} + +/// The size and level count of a bound texture, for validation without a device. +internal readonly record struct ComputeImageInfo(uint Width, uint Height, uint MipLevels, uint Layers); + +/// +/// The device-free half of a compute pass: which usage each binding is, whether the +/// bindings are consistent, how many groups cover an image, and the signature the +/// frame plan sees. +/// +internal static class ComputePassPlanner +{ + public static ResourceUsage UsageOf(ComputeAccess access) => access switch + { + ComputeAccess.Sampled => ResourceUsage.SampleCompute, + ComputeAccess.StorageRead => ResourceUsage.StorageReadCompute, + ComputeAccess.StorageWrite => ResourceUsage.StorageWrite, + ComputeAccess.StorageReadWrite => ResourceUsage.StorageReadWrite, + _ => throw new ArgumentOutOfRangeException(nameof(access), access, null), + }; + + public static bool IsStorage(ComputeAccess access) => access != ComputeAccess.Sampled; + + public static bool Writes(ComputeAccess access) => + access is ComputeAccess.StorageWrite or ComputeAccess.StorageReadWrite; + + /// The extent of one level: the base extent halved per level, never below 1. + public static uint LevelExtent(uint extent, uint mip) => Math.Max(1u, mip >= 32 ? 1u : extent >> (int)mip); + + /// Work groups of that cover texels. + public static uint GroupsCovering(uint extent, uint localSize) => + localSize == 0 ? 0 : (extent + localSize - 1) / localSize; + + /// + /// Why the declaration cannot be recorded, or null. Checks every binding names a + /// live single-layer texture and levels inside it, that storage bindings name one + /// level, that no binding number repeats, that no level is bound twice where one + /// of the uses writes (one layout per level per pass), and that every dispatch + /// sizes from an existing binding. + /// + public static string? Validate(ComputePassDeclaration pass, Func image) + { + if (pass.Dispatches.Length == 0) return "compute pass '" + pass.Name + "' has no dispatch"; + + var numbers = new HashSet(); + for (int i = 0; i < pass.Bindings.Length; i++) + { + ComputeBinding binding = pass.Bindings[i]; + if (!numbers.Add(binding.Binding)) return "binding " + binding.Binding + " is bound twice"; + + ComputeImageInfo? info = image(binding.TextureId); + if (info == null) return "binding " + binding.Binding + " names no texture (" + binding.TextureId + ")"; + if (info.Value.Layers > 1) return "binding " + binding.Binding + " names a layered texture"; + if (binding.MipCount == 0) return "binding " + binding.Binding + " covers no level"; + if (IsStorage(binding.Access) && binding.MipCount != 1) + return "storage binding " + binding.Binding + " must name exactly one level"; + if (binding.BaseMip + binding.MipCount > info.Value.MipLevels) + return "binding " + binding.Binding + " names levels " + binding.BaseMip + "+" + binding.MipCount + + " of a " + info.Value.MipLevels + "-level texture"; + + for (int j = 0; j < i; j++) + { + ComputeBinding other = pass.Bindings[j]; + if (other.TextureId != binding.TextureId) continue; + bool overlap = binding.BaseMip < other.BaseMip + other.MipCount && + other.BaseMip < binding.BaseMip + binding.MipCount; + if (!overlap) continue; + if (UsageOf(binding.Access) != UsageOf(other.Access) || Writes(binding.Access)) + return "bindings " + other.Binding + " and " + binding.Binding + + " use one level of texture " + binding.TextureId + " in two ways"; + } + } + + foreach (ComputeDispatch dispatch in pass.Dispatches) + { + if (dispatch.SizeFromBinding >= pass.Bindings.Length) + return "a dispatch sizes from binding index " + dispatch.SizeFromBinding + " of " + pass.Bindings.Length; + } + return null; + } + + /// The group counts of . + public static (uint X, uint Y, uint Z) Groups(ComputeDispatch dispatch, ComputePassDeclaration pass, + Func image, uint localSizeX, uint localSizeY) + { + if (dispatch.SizeFromBinding < 0) return (dispatch.GroupsX, dispatch.GroupsY, dispatch.GroupsZ); + ComputeBinding binding = pass.Bindings[dispatch.SizeFromBinding]; + ComputeImageInfo info = image(binding.TextureId) ?? default; + return (GroupsCovering(LevelExtent(info.Width, binding.BaseMip), localSizeX), + GroupsCovering(LevelExtent(info.Height, binding.BaseMip), localSizeY), 1); + } + + /// + /// What the frame plan sees of the pass: every written texture as a non-transient + /// attachment use (so a raster pass that attaches it later never loads DONT_CARE + /// over the dispatch's result) and every read texture in . + /// The extent is the first written level's, or the first binding's. + /// + public static PassSignature Signature(int nameId, ComputePassDeclaration pass, Func image) + { + var uses = new List(); + var reads = new List(); + uint width = 0, height = 0; + foreach (ComputeBinding binding in pass.Bindings) + { + if (Writes(binding.Access)) + { + var use = new AttachmentUse(binding.TextureId, UsageOf(binding.Access), false); + if (!uses.Contains(use)) uses.Add(use); + if (width == 0 && image(binding.TextureId) is { } written) + { + width = LevelExtent(written.Width, binding.BaseMip); + height = LevelExtent(written.Height, binding.BaseMip); + } + } + if (binding.Access != ComputeAccess.StorageWrite && !reads.Contains(binding.TextureId)) + { + reads.Add(binding.TextureId); + } + } + if (width == 0 && pass.Bindings.Length > 0 && image(pass.Bindings[0].TextureId) is { } first) + { + width = LevelExtent(first.Width, pass.Bindings[0].BaseMip); + height = LevelExtent(first.Height, pass.Bindings[0].BaseMip); + } + + return new PassSignature + { + NameId = nameId, + Attachments = uses.ToArray(), + Reads = reads.ToArray(), + Width = (int)width, + Height = (int)height, + FormatsId = -1, + }; + } +} diff --git a/Optimum.Render.Vulkan/Graph/FeedbackCopyPool.cs b/Optimum.Render.Vulkan/Graph/FeedbackCopyPool.cs new file mode 100644 index 00000000..fe2e8011 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/FeedbackCopyPool.cs @@ -0,0 +1,158 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Graph; + +/// The shape of a ReadSelf copy: copies with equal descriptions are interchangeable. +internal readonly record struct FeedbackCopyDesc(uint Width, uint Height, Format Format, uint MipLevels, uint Layers, bool Cube); + +/// +/// Pooled ReadSelf copies: the snapshot a draw samples when it reads a colour attachment +/// it is also writing (atlas composition). Replaces the permanent copy per source texture. +/// +/// A pass takes a copy () and gives it back when it ends +/// (). Within the frame being recorded a released copy is reused at +/// once: the commands that sampled it come earlier in the same command stream. A copy +/// released in an earlier frame is reused only after the Frame timeline completed the value +/// recorded when that frame ended (, ), and a +/// copy that stays free for frames is destroyed (the device +/// retires it on the timeline). +/// +internal sealed class FeedbackCopyPool +{ + private sealed class Copy + { + public int TextureId; + public FeedbackCopyDesc Desc; + public ulong RetiredAt; + public long FreeSince; + } + + private readonly ITimelineClock _clock; + private readonly Func _create; + private readonly Action _destroy; + private readonly Dictionary _inUse = new(); + private readonly List _releasedThisFrame = new(); + private readonly List _retiring = new(); + private readonly List _free = new(); + private long _frame; + + public FeedbackCopyPool(ITimelineClock clock, Func create, Action destroy, + int idleFrames = 120) + { + _clock = clock ?? throw new ArgumentNullException(nameof(clock)); + _create = create ?? throw new ArgumentNullException(nameof(create)); + _destroy = destroy ?? throw new ArgumentNullException(nameof(destroy)); + IdleFrames = idleFrames; + } + + public int IdleFrames { get; } + + /// Copies taken and not yet released. + public int InUse => _inUse.Count; + + /// Copies released in earlier frames whose timeline value has not completed. + public int Retiring => _retiring.Count; + + /// Copies ready for any frame. + public int Free => _free.Count + _releasedThisFrame.Count; + + /// Every copy the pool holds. + public int Live => _inUse.Count + _releasedThisFrame.Count + _retiring.Count + _free.Count; + + /// Copies created so far. + public long Created { get; private set; } + + /// Copies destroyed so far. + public long Destroyed { get; private set; } + + /// A copy of for one pass; its texture id. + public int Acquire(FeedbackCopyDesc desc) + { + Copy? copy = Take(_releasedThisFrame, desc) ?? Take(_free, desc); + if (copy == null) + { + copy = new Copy { TextureId = _create(desc), Desc = desc }; + Created++; + } + _inUse.Add(copy.TextureId, copy); + return copy.TextureId; + } + + /// Gives a copy back at the end of its pass. Unknown ids are ignored. + public void Release(int copyId) + { + if (_inUse.Remove(copyId, out Copy? copy)) _releasedThisFrame.Add(copy); + } + + /// + /// Ends the recorded frame: its released copies wait for the Frame value recorded now. + /// Call before the next frame reserves its value. + /// + public void EndFrame() + { + if (_releasedThisFrame.Count == 0) return; + ulong recorded = _clock.FrameRecorded; + foreach (Copy copy in _releasedThisFrame) + { + copy.RetiredAt = recorded; + _retiring.Add(copy); + } + _releasedThisFrame.Clear(); + } + + /// + /// Frees the copies whose timeline value completed and destroys copies free for more + /// than frames. Call once per frame after . + /// + public void Collect() + { + _frame++; + ulong completed = _clock.FrameCompleted; + int kept = 0; + for (int i = 0; i < _retiring.Count; i++) + { + Copy copy = _retiring[i]; + if (copy.RetiredAt <= completed) + { + copy.FreeSince = _frame; + _free.Add(copy); + } + else + { + _retiring[kept++] = copy; + } + } + _retiring.RemoveRange(kept, _retiring.Count - kept); + + kept = 0; + for (int i = 0; i < _free.Count; i++) + { + Copy copy = _free[i]; + if (_frame - copy.FreeSince > IdleFrames) + { + _destroy(copy.TextureId); + Destroyed++; + } + else + { + _free[kept++] = copy; + } + } + _free.RemoveRange(kept, _free.Count - kept); + } + + private static Copy? Take(List list, FeedbackCopyDesc desc) + { + for (int i = list.Count - 1; i >= 0; i--) + { + if (list[i].Desc != desc) continue; + Copy copy = list[i]; + list.RemoveAt(i); + return copy; + } + return null; + } +} diff --git a/Optimum.Render.Vulkan/Graph/FrameGraph.cs b/Optimum.Render.Vulkan/Graph/FrameGraph.cs new file mode 100644 index 00000000..1168359a --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/FrameGraph.cs @@ -0,0 +1,361 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Graph; + +/// How a declared pass treats sampling and scope splits. +[Flags] +public enum PassFlags +{ + None = 0, + + /// + /// Mod-hosted stages sample what they like: at pass entry every render-target + /// texture outside the pass's attachments that is not already shader-readable + /// moves to SHADER_READ_ONLY_OPTIMAL, so an undeclared read does not split. + /// + OpenSampling = 1, + + /// + /// A split (a second vkCmdBeginRendering inside the pass) is expected here and + /// is not traced as a declaration violation. It is still counted. + /// + AllowSplit = 2, +} + +/// +/// One pass as the platform declares it: the target, which of its colour slots +/// take part (the rest are null attachments, so they can be sampled), what it +/// reads, which slots it overwrites completely every frame (transient for the +/// plan), and its sampling policy. Depth follows the target: the scope holds the +/// bound depth attachment, writable or read-only as the draws need. +/// +public sealed class PassDeclaration +{ + /// The bound framebuffer; for the default target. + public const int BoundFramebuffer = 0; + + public const int DefaultFramebuffer = -1; + + public string Name = ""; + + /// Render target id; declares on whatever is bound. + public int FramebufferId = BoundFramebuffer; + + /// Bit i: colour slot i is an attachment of the pass. Default: every bound slot. + public uint ColorSlots = uint.MaxValue; + + /// Texture ids the pass samples (not its own attachments), pre-transitioned at pass entry. + public int[] Reads = Array.Empty(); + + /// + /// Bit i: colour slot i is plainly written over its whole extent by the pass + /// and never needed from a previous frame, so an exactly matching plan may + /// load it DONT_CARE. + /// + public uint TransientSlots; + + public PassFlags Flags; +} + +/// A clear issued with no pass open, waiting for the next use of its image. +internal readonly record struct PendingClear(VulkanTexture Texture, uint Layer, bool Depth, float R, float G, float B, float A); + +/// +/// The streaming frame graph (Vulkan-native plan, Phase 2 step 2). The client's +/// frame is imperative, so passes are declared and recorded in frame order as they +/// happen: asks for a pass index when a pass +/// opens its one rendering scope (), and the load ops come +/// from the solved from the previous frame, applied only +/// while every pass so far matches it exactly. Store ops stay STORE: a streaming +/// recorder cannot know that the rest of the frame will still match (see +/// ). +/// +/// It also holds the clears issued with no pass open (clear promotion): they become +/// LOAD_OP_CLEAR on the next scope that attaches the image, or a standalone clear +/// command when something else touches the image first. +/// +/// OPTIMUM_VULKAN_FRAMEGRAPH=0 turns all of it off and keeps the scope +/// inference path exactly as it was. Render thread only. +/// +internal sealed class FrameGraph +{ + public const string Variable = "OPTIMUM_VULKAN_FRAMEGRAPH"; + + public static bool EnabledByEnvironment => Environment.GetEnvironmentVariable(Variable) != "0"; + + /// Change only between frames. + public bool Enabled { get; set; } = EnabledByEnvironment; + + private readonly Dictionary _names = new(StringComparer.Ordinal); + private readonly List _frame = new(); + private readonly List _pending = new(); + + // The plans of the last two frames: the TAA history ping-pong makes every frame's + // signature differ from the one before it but equal to the one before that. + private readonly FramePlan?[] _plans = new FramePlan?[2]; + private readonly bool[] _prefixMatches = { true, true }; + + // Totals for tests; VulkanStats carries the interval counters. + public long Passes { get; private set; } + public long DeclaredPasses { get; private set; } + public long Splits { get; private set; } + public long UndeclaredSplits { get; private set; } + public long PlanHits { get; private set; } + public long PlanMisses { get; private set; } + public long InPassClears { get; private set; } + public long PromotedClears { get; private set; } + public long StandaloneClears { get; private set; } + public long PlannedDontCareLoads { get; private set; } + public long ComputePasses { get; private set; } + public long Dispatches { get; private set; } + + /// Passes opened in the frame being recorded. + public int PassesThisFrame => _frame.Count; + + /// The plan solved from the previous frame, null before the first frame ended. + public FramePlan? Plan => _plans[0]; + + /// Whether every pass opened so far this frame matches one of the last two plans. + public bool PrefixMatchesPlan => _prefixMatches[0] || _prefixMatches[1]; + + public int NameId(string name) + { + if (!_names.TryGetValue(name, out int id)) + { + id = _names.Count; + _names.Add(name, id); + } + return id; + } + + /// + /// A pass opens its scope: records its signature and returns its index in the + /// frame. is false for a scope no declaration + /// covered (the inference path inside a graph frame). + /// + public int OpenPass(PassSignature signature, bool declared) + { + int index = RecordPass(signature); + Passes++; + if (declared) DeclaredPasses++; + VulkanStats.NotePass(); + if (RenderTrace.Enabled) + { + RenderTrace.Write("pass " + index + " name=" + signature.NameId + " attachments=" + signature.Attachments.Length + + " reads=" + signature.Reads.Length + " " + signature.Width + "x" + signature.Height + + " plan=" + (PrefixMatchesPlan ? "match" : "conservative")); + } + return index; + } + + /// + /// A compute pass was recorded: counted, and, while the graph is on, its signature + /// (storage writes as attachment uses, reads as reads) joins the frame so the plan + /// sees the dispatch's writes and reads between the raster passes around it. It opens + /// no scope, so it is not one of . + /// + public int OpenComputePass(PassSignature signature) + { + ComputePasses++; + VulkanStats.NoteComputePass(); + if (!Enabled) return -1; + + int index = RecordPass(signature); + if (RenderTrace.Enabled) + { + RenderTrace.Write("compute pass " + index + " name=" + signature.NameId + " writes=" + + signature.Attachments.Length + " reads=" + signature.Reads.Length + " " + signature.Width + "x" + + signature.Height + " plan=" + (PrefixMatchesPlan ? "match" : "conservative")); + } + return index; + } + + private int RecordPass(PassSignature signature) + { + int index = _frame.Count; + for (int k = 0; k < _plans.Length; k++) + { + FramePlan? plan = _plans[k]; + _prefixMatches[k] = _prefixMatches[k] && plan != null && + plan.MatchesPass(index, signature); + } + _frame.Add(signature); + return index; + } + + public void NoteDispatch() + { + Dispatches++; + VulkanStats.NoteDispatch(); + } + + /// + /// The load op for attachment of pass + /// when no clear was promoted into it: the plan's op + /// while the frame so far matches the plan, LOAD otherwise. + /// + public AttachmentLoadOp PlannedLoad(int pass, int attachment) + { + FramePlan? plan = _prefixMatches[0] ? _plans[0] : _prefixMatches[1] ? _plans[1] : null; + if (plan == null || pass < 0 || pass >= plan.PassCount) return AttachmentLoadOp.Load; + AttachmentLoadOp op = plan.LoadOp(pass, attachment); + if (op == AttachmentLoadOp.DontCare) PlannedDontCareLoads++; + return op; + } + + /// A scope reopened inside a pass that had already opened one. + public void NoteSplit(bool allowed) + { + Splits++; + if (!allowed) UndeclaredSplits++; + VulkanStats.NotePassSplit(); + } + + public void NoteInPassClear() + { + InPassClears++; + VulkanStats.NoteInPassClear(); + } + + /// + /// Ends the frame: a hit when every pass matched the plan the frame was recorded + /// against. Reuses a matching plan or solves a new one when the frame changed. + /// + public void EndFrame() + { + if (_frame.Count > 0) + { + int match = -1; + for (int i = 0; i < _plans.Length; i++) + { + FramePlan? plan = _plans[i]; + if (plan != null && plan.Matches(_frame)) + { + match = i; + break; + } + } + if (match >= 0) + { + PlanHits++; + VulkanStats.NotePlanHit(); + // Keep both recurring frame shapes (e.g. alternating TAA history). + // A matching immutable plan needs no new snapshots or load-op solve. + if (match == 1) (_plans[0], _plans[1]) = (_plans[1], _plans[0]); + } + else + { + PlanMisses++; + VulkanStats.NotePlanMiss(); + _plans[1] = _plans[0]; + _plans[0] = FramePlan.Build(_frame); + } + } + _frame.Clear(); + _prefixMatches[0] = true; + _prefixMatches[1] = true; + } + + // ------------------------------------------------------------ clear promotion + + public bool HasPendingClears => _pending.Count > 0; + + public void PromoteColorClear(VulkanTexture texture, uint layer, float r, float g, float b, float a) + { + Replace(new PendingClear(texture, layer, false, r, g, b, a)); + } + + public void PromoteDepthClear(VulkanTexture texture, float depth) + { + Replace(new PendingClear(texture, 0, true, depth, 0, 0, 0)); + } + + private void Replace(PendingClear clear) + { + for (int i = 0; i < _pending.Count; i++) + { + PendingClear existing = _pending[i]; + if (ReferenceEquals(existing.Texture, clear.Texture) && existing.Layer == clear.Layer && + existing.Depth == clear.Depth) + { + _pending[i] = clear; + return; + } + } + _pending.Add(clear); + } + + public bool HasPendingClear(VulkanTexture texture) + { + for (int i = 0; i < _pending.Count; i++) + { + if (ReferenceEquals(_pending[i].Texture, texture)) return true; + } + return false; + } + + /// Takes the clear of one attachment view into a load op, if one is pending. + public bool TakeForLoad(VulkanTexture texture, uint layer, bool depth, out PendingClear clear) + { + for (int i = 0; i < _pending.Count; i++) + { + PendingClear candidate = _pending[i]; + if (!ReferenceEquals(candidate.Texture, texture) || candidate.Depth != depth) continue; + if (!depth && candidate.Layer != layer) continue; + _pending.RemoveAt(i); + clear = candidate; + PromotedClears++; + VulkanStats.NotePromotedClear(); + return true; + } + clear = default; + return false; + } + + /// Takes every clear pending on (null: on every texture) for standalone commands. + public void TakeStandalone(VulkanTexture? texture, List output) + { + for (int i = _pending.Count - 1; i >= 0; i--) + { + if (texture != null && !ReferenceEquals(_pending[i].Texture, texture)) continue; + output.Add(_pending[i]); + _pending.RemoveAt(i); + } + // Oldest first, so two clears never reorder. + output.Reverse(); + } + + public void NoteStandaloneClear() + { + StandaloneClears++; + VulkanStats.NoteStandaloneClear(); + } + + /// A deleted texture's clears are moot. + public void Drop(VulkanTexture texture) + { + for (int i = _pending.Count - 1; i >= 0; i--) + { + if (ReferenceEquals(_pending[i].Texture, texture)) _pending.RemoveAt(i); + } + } +} + +/// +/// Vulkan-native plan, Phase 2 (contract C3): receives the render-stage bracket +/// ClientMain.TriggerRenderStage issues around each stage's renderers, forwarded by +/// . The frame graph implements it to map +/// (stage, target) to passes. Called on the render thread only. +/// +internal interface IRenderStageListener +{ + /// Before the stage's renderers run. + void OnBeginRenderStage(EnumRenderStage stage); + + /// After the stage's renderers ran, before the GL error check. + void OnEndRenderStage(EnumRenderStage stage); +} diff --git a/Optimum.Render.Vulkan/Graph/FramePlan.cs b/Optimum.Render.Vulkan/Graph/FramePlan.cs new file mode 100644 index 00000000..31ccfc31 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/FramePlan.cs @@ -0,0 +1,185 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Graph; + +/// +/// Cached attachment load operations for a recorded frame. The streaming graph applies +/// them only while the current pass prefix matches. Stores always preserve contents; +/// physical image reuse belongs to TransientAllocator and its explicit lifetimes. +/// +internal sealed class FramePlan +{ + private readonly PassSignature[] _passes; + private readonly AttachmentLoadOp[][] _load; + + private FramePlan(PassSignature[] passes, AttachmentLoadOp[][] load) + { + _passes = passes; + _load = load; + } + + public int PassCount => _passes.Length; + + /// Snapshot the frame and identify attachments whose initial contents + /// are disposable. Every attachment use must opt in, and the first reference + /// must be a plain write with no read of that resource in the same pass. + public static FramePlan Build(IReadOnlyList frame) + { + PassSignature[] passes = Snapshot(frame); + var resources = new Dictionary(); + for (int p = 0; p < passes.Length; p++) + { + foreach (AttachmentUse use in passes[p].Attachments) + { + ResourceInfo info = Touch(resources, use.ResourceId, p); + if (!use.Transient || (info.FirstPass == p && !IsPlainWrite(use.Usage))) + info.DiscardInitialContents = false; + } + foreach (int read in passes[p].Reads) + { + ResourceInfo info = Touch(resources, read, p); + if (info.FirstPass == p) info.DiscardInitialContents = false; + } + } + + var load = new AttachmentLoadOp[passes.Length][]; + for (int p = 0; p < passes.Length; p++) + { + AttachmentUse[] attachments = passes[p].Attachments; + load[p] = new AttachmentLoadOp[attachments.Length]; + for (int a = 0; a < attachments.Length; a++) + { + AttachmentUse use = attachments[a]; + ResourceInfo info = resources[use.ResourceId]; + load[p][a] = info.DiscardInitialContents && info.FirstPass == p && IsPlainWrite(use.Usage) + ? AttachmentLoadOp.DontCare + : AttachmentLoadOp.Load; + } + } + return new FramePlan(passes, load); + } + + /// Exact match on pass count and, per pass, name, attachments (resource, usage, + /// transient flag, order), reads (order significant), extent and formats. + public bool Matches(IReadOnlyList frame) + { + if (frame == null || frame.Count != _passes.Length) return false; + for (int p = 0; p < _passes.Length; p++) + { + if (!_passes[p].SameAs(frame[p])) return false; + } + return true; + } + + /// Whether pass of this plan is identical to + /// . False for an index past the plan's end. + public bool MatchesPass(int pass, PassSignature signature) + { + if (pass < 0 || pass >= _passes.Length) return false; + return _passes[pass].SameAs(signature); + } + + public AttachmentLoadOp LoadOp(int pass, int attachment) + { + CheckIndex(pass, attachment); + return _load[pass][attachment]; + } + + private void CheckIndex(int pass, int attachment) + { + if ((uint)pass >= (uint)_passes.Length) + throw new ArgumentOutOfRangeException(nameof(pass), pass, $"Plan has {_passes.Length} passes."); + if ((uint)attachment >= (uint)_load[pass].Length) + throw new ArgumentOutOfRangeException(nameof(attachment), attachment, $"Pass {pass} has {_load[pass].Length} attachments."); + } + + private static bool IsPlainWrite(ResourceUsage usage) => + usage == ResourceUsage.ColorWrite || usage == ResourceUsage.DepthWrite; + + private static PassSignature[] Snapshot(IReadOnlyList frame) + { + if (frame == null) throw new ArgumentNullException(nameof(frame)); + var passes = new PassSignature[frame.Count]; + for (int p = 0; p < passes.Length; p++) + { + PassSignature source = frame[p] ?? throw new ArgumentException($"Pass {p} is null.", nameof(frame)); + passes[p] = source.Clone(); + } + return passes; + } + + private static ResourceInfo Touch(Dictionary resources, int resourceId, int pass) + { + if (!resources.TryGetValue(resourceId, out ResourceInfo? info)) + { + info = new ResourceInfo(pass); + resources.Add(resourceId, info); + } + return info; + } + + private sealed class ResourceInfo + { + public ResourceInfo(int firstPass) => FirstPass = firstPass; + public readonly int FirstPass; + public bool DiscardInitialContents = true; + } +} + +/// One attachment of a pass: which resource, how it is used, and whether its +/// contents are allowed to die with the frame. +/// The contents are never read in a later frame. A resource is +/// treated as transient only when every attachment use of it in the frame says so and its +/// first reference is a plain write (see ). +internal readonly record struct AttachmentUse(int ResourceId, ResourceUsage Usage, bool Transient); + +/// +/// What a pass looked like when it was recorded: the unit a is +/// built from and matched against. Resource ids are frame-graph ids, stable across frames +/// for the same logical image. +/// +internal sealed class PassSignature +{ + public int NameId; + public AttachmentUse[] Attachments = Array.Empty(); + /// Resources sampled or otherwise read (not as attachments), in declaration order. + public int[] Reads = Array.Empty(); + public int Width; + public int Height; + /// Interned id of the ordered attachment format list. + public int FormatsId; + + public PassSignature Clone() => new() + { + NameId = NameId, + Attachments = Attachments == null ? Array.Empty() : (AttachmentUse[])Attachments.Clone(), + Reads = Reads == null ? Array.Empty() : (int[])Reads.Clone(), + Width = Width, + Height = Height, + FormatsId = FormatsId, + }; + + /// Exact equality on every field the plan depends on. Null arrays equal empty ones; + /// read order is significant. + public bool SameAs(PassSignature other) + { + if (other == null) return false; + if (NameId != other.NameId || Width != other.Width || Height != other.Height || FormatsId != other.FormatsId) + return false; + return SameSequence(Attachments, other.Attachments) && SameSequence(Reads, other.Reads); + } + + private static bool SameSequence(T[]? a, T[]? b) where T : IEquatable + { + int la = a?.Length ?? 0; + int lb = b?.Length ?? 0; + if (la != lb) return false; + for (int i = 0; i < la; i++) + { + if (!a![i].Equals(b![i])) return false; + } + return true; + } +} diff --git a/Optimum.Render.Vulkan/Graph/PassRecorder.cs b/Optimum.Render.Vulkan/Graph/PassRecorder.cs new file mode 100644 index 00000000..4e8f10e0 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/PassRecorder.cs @@ -0,0 +1,280 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Graph; + +/// +/// Opens the one rendering scope of a pass (Vulkan-native plan, Phase 2 step 2). +/// +/// owns the bound target and decides when a scope +/// has to open; on the frame-graph path it hands the attachment set to +/// , which, before vkCmdBeginRendering and outside any +/// scope: +/// +/// records the standalone clears a read or a read-only depth attachment needs +/// (a clear promoted into an image that is read before any pass attaches it); +/// queues the pass-entry barriers: every declared read (or, for an +/// pass, every render-target texture outside +/// the pass) to SHADER_READ_ONLY, every attachment to its attachment usage; +/// records the pass signature with (a second scope +/// in one declared pass is a split, counted, not a new pass); +/// chooses each attachment's load op: CLEAR when a clear was promoted into +/// it, otherwise the plan's op on the pass's first scope, LOAD on a split. +/// +/// The caller flushes the batcher once and begins rendering, so every barrier of +/// the pass is one vkCmdPipelineBarrier2. +/// +internal sealed unsafe class PassRecorder +{ + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly BarrierBatcher _barriers; + private readonly FrameGraph _graph; + private readonly List _clears = new(); + private readonly List _uses = new(); + + public PassRecorder(VulkanContext context, TextureManager textures, BarrierBatcher barriers, FrameGraph graph) + { + _context = context; + _textures = textures; + _barriers = barriers; + _graph = graph; + } + + public FrameGraph Graph => _graph; + + /// The declared pass, while one is current. + public PassDeclaration? Declared { get; private set; } + + /// The target the declared pass was declared on. + public VulkanFramebuffer? DeclaredOn { get; private set; } + + /// Whether the declared pass has opened its scope. + public bool Opened { get; private set; } + + public void Declare(PassDeclaration declaration, VulkanFramebuffer framebuffer) + { + Declared = declaration; + DeclaredOn = framebuffer; + Opened = false; + } + + public void ClearDeclaration() + { + Declared = null; + DeclaredOn = null; + Opened = false; + } + + /// + /// Records every clear pending on (null: on every + /// texture) as a clear-image command. No rendering scope may be open. + /// + public void FlushClears(CommandBuffer commandBuffer, VulkanTexture? texture) + { + if (!_graph.HasPendingClears) return; + _clears.Clear(); + _graph.TakeStandalone(texture, _clears); + foreach (PendingClear clear in _clears) + { + VulkanTexture target = clear.Texture; + _textures.Require(_barriers, commandBuffer, target, ResourceUsage.TransferDst); + _barriers.Flush(commandBuffer); + if (clear.Depth) + { + var value = new ClearDepthStencilValue(clear.R, 0); + var range = new ImageSubresourceRange(target.Aspect, 0, 1, 0, 1); + _context.Api.CmdClearDepthStencilImage(commandBuffer, target.Image, ImageLayout.TransferDstOptimal, + &value, 1, &range); + } + else + { + var value = new ClearColorValue(clear.R, clear.G, clear.B, clear.A); + var range = new ImageSubresourceRange(ImageAspectFlags.ColorBit, 0, 1, clear.Layer, 1); + _context.Api.CmdClearColorImage(commandBuffer, target.Image, ImageLayout.TransferDstOptimal, + &value, 1, &range); + } + _graph.NoteStandaloneClear(); + if (RenderTrace.Enabled) + { + RenderTrace.Write("standalone clear image=" + target.Image.Handle.ToString("x") + + (clear.Depth ? " depth=" + clear.R : " layer=" + clear.Layer + " rgba=" + clear.R + "," + clear.G + + "," + clear.B + "," + clear.A)); + } + } + _clears.Clear(); + } + + /// + /// Queues the barriers and chooses the load ops of the scope about to open on + /// . holds the texture + /// of every slot in the scope (null for a slot left out) and is filled into + /// ' load ops and clear values; the views and + /// layouts are the caller's. + /// + public void Prepare(CommandBuffer commandBuffer, VulkanFramebuffer framebuffer, ReadOnlySpan colour, + VulkanTexture? depth, bool depthReadOnly, int formatsId, IReadOnlyList framebuffers, + Span attachments, ref RenderingAttachmentInfo depthAttachment) + { + PassDeclaration? declaration = ReferenceEquals(DeclaredOn, framebuffer) ? Declared : null; + bool split = declaration != null && Opened; + + // 1. Clears promoted into images this pass reads: they must land before the read. + if (_graph.HasPendingClears) + { + if (declaration != null) + { + foreach (int id in declaration.Reads) + { + VulkanTexture? read = _textures.Get(id); + if (read != null && !InScope(read, colour, depth)) FlushClears(commandBuffer, read); + } + if ((declaration.Flags & PassFlags.OpenSampling) != 0) + { + ForEachOpenSamplingCandidate(commandBuffer, framebuffers, colour, depth, flushClears: true); + } + } + // A read-only depth attachment cannot take LOAD_OP_CLEAR. + if (depth != null && depthReadOnly) FlushClears(commandBuffer, depth); + } + + // 2. Pass-entry barriers for the reads. + if (declaration != null) + { + foreach (int id in declaration.Reads) + { + VulkanTexture? read = _textures.Get(id); + if (read == null || InScope(read, colour, depth)) continue; + _textures.Require(_barriers, commandBuffer, read, ResourceUsage.SampleFragment); + } + if ((declaration.Flags & PassFlags.OpenSampling) != 0) + { + ForEachOpenSamplingCandidate(commandBuffer, framebuffers, colour, depth, flushClears: false); + } + } + + // 3. The signature, and the pass index the plan is consulted with. + uint transient = declaration?.TransientSlots ?? 0; + _uses.Clear(); + for (int i = 0; i < colour.Length; i++) + { + if (colour[i] == null) continue; + bool isTransient = ((transient >> i) & 1) != 0; + _uses.Add(new AttachmentUse(framebuffer.Color[i].TextureId, + isTransient ? ResourceUsage.ColorWrite : ResourceUsage.ColorBlend, isTransient)); + } + if (depth != null) + { + _uses.Add(new AttachmentUse(framebuffer.DepthTextureId, + depthReadOnly ? ResourceUsage.DepthReadOnlySampled : ResourceUsage.DepthWrite, false)); + } + + int passIndex; + if (split) + { + passIndex = -1; + bool allowed = (declaration!.Flags & PassFlags.AllowSplit) != 0; + _graph.NoteSplit(allowed); + if (!allowed && RenderTrace.Enabled) + { + RenderTrace.Write("pass split: '" + declaration.Name + "' reopened its scope on framebuffer " + + framebuffer.Id); + } + } + else + { + var signature = new PassSignature + { + NameId = _graph.NameId(declaration?.Name ?? "~implicit"), + Attachments = _uses.ToArray(), + Reads = declaration != null ? (int[])declaration.Reads.Clone() : Array.Empty(), + Width = (int)framebuffer.Width, + Height = (int)framebuffer.Height, + FormatsId = formatsId, + }; + passIndex = _graph.OpenPass(signature, declaration != null); + if (declaration != null) Opened = true; + } + + // 4. Attachment barriers and load ops. + int use = 0; + for (int i = 0; i < colour.Length; i++) + { + VulkanTexture? texture = colour[i]; + if (texture == null) continue; + + AttachmentUse attachment = _uses[use]; + if (_graph.TakeForLoad(texture, framebuffer.Color[i].Layer, depth: false, out PendingClear clear)) + { + attachments[i].LoadOp = AttachmentLoadOp.Clear; + attachments[i].ClearValue = new ClearValue(new ClearColorValue(clear.R, clear.G, clear.B, clear.A)); + } + else + { + attachments[i].LoadOp = passIndex >= 0 ? _graph.PlannedLoad(passIndex, use) : AttachmentLoadOp.Load; + } + // LOAD reads the attachment: a plain-write declaration only drops the read + // access when the contents are not loaded (sync validation: read-after-write). + ResourceUsage usage = attachment.Usage == ResourceUsage.ColorWrite && attachments[i].LoadOp == AttachmentLoadOp.Load + ? ResourceUsage.ColorBlend + : attachment.Usage; + _textures.Require(_barriers, commandBuffer, texture, usage); + use++; + } + + if (depth != null) + { + _textures.Require(_barriers, commandBuffer, depth, _uses[use].Usage); + if (!depthReadOnly && _graph.TakeForLoad(depth, 0, depth: true, out PendingClear clear)) + { + depthAttachment.LoadOp = AttachmentLoadOp.Clear; + depthAttachment.ClearValue = new ClearValue(depthStencil: new ClearDepthStencilValue(clear.R, 0)); + } + else + { + depthAttachment.LoadOp = passIndex >= 0 ? _graph.PlannedLoad(passIndex, use) : AttachmentLoadOp.Load; + } + } + } + + private static bool InScope(VulkanTexture texture, ReadOnlySpan colour, VulkanTexture? depth) + { + if (ReferenceEquals(texture, depth)) return true; + for (int i = 0; i < colour.Length; i++) + { + if (ReferenceEquals(colour[i], texture)) return true; + } + return false; + } + + /// + /// Every render-target texture outside the scope that is not already shader-readable: + /// what a mod-hosted stage might sample. + /// + private void ForEachOpenSamplingCandidate(CommandBuffer commandBuffer, IReadOnlyList framebuffers, + ReadOnlySpan colour, VulkanTexture? depth, bool flushClears) + { + for (int f = 0; f < framebuffers.Count; f++) + { + VulkanFramebuffer? other = framebuffers[f]; + if (other == null) continue; + for (int i = 0; i <= other.Color.Length; i++) + { + int id = i < other.Color.Length ? other.Color[i].TextureId : other.DepthTextureId; + if (id <= 0) continue; + VulkanTexture? texture = _textures.Get(id); + if (texture == null || InScope(texture, colour, depth)) continue; + if (flushClears) + { + FlushClears(commandBuffer, texture); + } + else if (texture.Layout != ImageLayout.ShaderReadOnlyOptimal) + { + _textures.Require(_barriers, commandBuffer, texture, ResourceUsage.SampleFragment); + } + } + } + } +} diff --git a/Optimum.Render.Vulkan/Graph/ResourceStateTracker.cs b/Optimum.Render.Vulkan/Graph/ResourceStateTracker.cs new file mode 100644 index 00000000..fa253836 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/ResourceStateTracker.cs @@ -0,0 +1,490 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Graph; + +/// +/// The synchronization state of one subresource (mip level, array layer). +/// +/// The layout the subresource is in. +/// The stage of the last write, or none. +/// The access of that write. +/// The stages the current contents were made visible to by a barrier. +/// The stages that read since the last barrier or write. +/// The accesses of those reads. +internal readonly record struct SubresourceState( + ImageLayout Layout, + PipelineStageFlags2 WriteStage, + AccessFlags2 WriteAccess, + PipelineStageFlags2 VisibleStages, + PipelineStageFlags2 ReadStages, + AccessFlags2 ReadAccess) +{ + public static SubresourceState Undefined => new(ImageLayout.Undefined, + PipelineStageFlags2.None, AccessFlags2.None, PipelineStageFlags2.None, PipelineStageFlags2.None, AccessFlags2.None); +} + +/// The two sides of one barrier, without its subresource range. +internal readonly record struct BarrierSides( + ImageLayout OldLayout, + ImageLayout NewLayout, + PipelineStageFlags2 SrcStage, + AccessFlags2 SrcAccess, + PipelineStageFlags2 DstStage, + AccessFlags2 DstAccess); + +/// One barrier over a rectangle of subresources. +internal readonly record struct ImageTransition(uint BaseMip, uint MipCount, uint BaseLayer, uint LayerCount, BarrierSides Sides); + +/// +/// Per-subresource layout, last write stage and access, the stages that write is +/// visible to, and the read stages since, for one image. Derives the barriers a +/// new use needs: +/// +/// a layout change always needs one; its source side names the last write +/// and the reads since (so the write is made available and the reads complete), +/// its destination side the new use; +/// read after write (RAW) in the same layout needs one only when the +/// reader's stage is not among the stages the write is visible to; +/// write after read (WAR) and write after write (WAW) in the same layout +/// need ordering even when the stages match; +/// read after read needs no barrier while layout and visibility agree. +/// +/// One entry covers the whole image while every subresource agrees; a use of a +/// sub-range splits it into per-subresource entries, and a use that makes them +/// agree again merges them back. Not thread-safe: +/// locks the tracker around each call. +/// +internal sealed class ResourceStateTracker +{ + private SubresourceState _whole = SubresourceState.Undefined; + private SubresourceState[]? _split; + + public ResourceStateTracker(uint mipLevels, uint layers, bool depth) + { + MipLevels = Math.Max(1, mipLevels); + Layers = Math.Max(1, layers); + Depth = depth; + } + + public uint MipLevels { get; } + public uint Layers { get; } + public bool Depth { get; } + + /// Whether sub-ranges currently differ (one entry per subresource). + public bool IsSplit => _split != null; + + /// The layout of the whole image, or UNDEFINED while sub-ranges differ. + public ImageLayout Layout => _split == null ? _whole.Layout : ImageLayout.Undefined; + + public SubresourceState StateOf(uint mip, uint layer) => + _split == null ? _whole : _split[mip * Layers + layer]; + + /// Forgets everything: the image's contents are undefined (a swapchain image before its frame). + public void Reset() => Reset(PipelineStageFlags2.None); + + /// + /// Forgets the contents, keeping one prior use: is + /// where something outside the command buffer last touched the image. A + /// swapchain image was accessed by vkAcquireNextImageKHR, and the submission + /// waits on the acquire semaphore at a stage, so the first barrier must name + /// that stage on its source side or synchronization validation reports + /// write-after-read against the acquire. + /// + public void Reset(PipelineStageFlags2 priorStage) + { + _whole = SubresourceState.Undefined with { VisibleStages = priorStage }; + _split = null; + } + + /// + /// The contents stop mattering (an aliased transient starts a new lifetime): the + /// next use of every subresource transitions from UNDEFINED, even into the layout + /// it is already in. The uses recorded so far stay, so that barrier's source side + /// still names them. + /// + public void Discard() + { + if (_split == null) + { + _whole = _whole with { Layout = ImageLayout.Undefined }; + return; + } + for (int i = 0; i < _split.Length; i++) _split[i] = _split[i] with { Layout = ImageLayout.Undefined }; + } + + /// + /// Records a use of a subresource range and appends the barriers it needs to + /// , as few rectangles as the state allows. + /// says the contents do not matter, so a layout + /// change starts from UNDEFINED. Returns how many were appended. + /// + public int Require(uint baseMip, uint mipCount, uint baseLayer, uint layerCount, ResourceUsage usage, + bool discard, List output) + { + if (baseMip >= MipLevels || baseLayer >= Layers) return 0; + mipCount = Math.Min(mipCount, MipLevels - baseMip); + layerCount = Math.Min(layerCount, Layers - baseLayer); + if (mipCount == 0 || layerCount == 0) return 0; + + UsageState target = UsageState.For(usage, Depth); + (PipelineStageFlags2 writeStage, AccessFlags2 writeAccess) = UsageState.WriteOf(usage, Depth); + + bool whole = baseMip == 0 && mipCount == MipLevels && baseLayer == 0 && layerCount == Layers; + if (_split == null && whole) + { + if (!Advance(ref _whole, target, writeStage, writeAccess, discard, out BarrierSides sides)) return 0; + output.Add(new ImageTransition(0, MipLevels, 0, Layers, sides)); + return 1; + } + + if (_split == null) + { + _split = new SubresourceState[MipLevels * Layers]; + Array.Fill(_split, _whole); + } + + int before = output.Count; + int firstRectangle = output.Count; + for (uint mip = baseMip; mip < baseMip + mipCount; mip++) + { + uint runStart = 0; + BarrierSides runSides = default; + bool inRun = false; + for (uint layer = baseLayer; layer < baseLayer + layerCount; layer++) + { + ref SubresourceState state = ref _split[mip * Layers + layer]; + bool needed = Advance(ref state, target, writeStage, writeAccess, discard, out BarrierSides sides); + if (inRun && (!needed || sides != runSides)) + { + AddRectangle(output, firstRectangle, mip, runStart, layer - runStart, runSides); + inRun = false; + } + if (needed && !inRun) + { + runStart = layer; + runSides = sides; + inRun = true; + } + } + if (inRun) AddRectangle(output, firstRectangle, mip, runStart, baseLayer + layerCount - runStart, runSides); + } + + TryMerge(); + return output.Count - before; + } + + /// + /// Adds a one-mip rectangle, extending the rectangle of the previous mip with + /// the same layer span and sides when there is one. + /// + private static void AddRectangle(List output, int first, uint mip, uint baseLayer, + uint layerCount, BarrierSides sides) + { + for (int i = first; i < output.Count; i++) + { + ImageTransition candidate = output[i]; + if (candidate.BaseLayer == baseLayer && candidate.LayerCount == layerCount && + candidate.BaseMip + candidate.MipCount == mip && candidate.Sides == sides) + { + output[i] = candidate with { MipCount = candidate.MipCount + 1 }; + return; + } + } + output.Add(new ImageTransition(mip, 1, baseLayer, layerCount, sides)); + } + + private void TryMerge() + { + if (_split == null) return; + SubresourceState first = _split[0]; + for (int i = 1; i < _split.Length; i++) + { + if (_split[i] != first) return; + } + _whole = first; + _split = null; + } + + /// + /// Applies one use to one subresource. Returns whether a barrier is needed + /// before it, and that barrier's sides. + /// + internal static bool Advance(ref SubresourceState state, UsageState target, + PipelineStageFlags2 writeStage, AccessFlags2 writeAccess, bool discard, out BarrierSides sides) + { + AccessFlags2 readAccess = target.Access & ~UsageState.WriteAccessMask; + PipelineStageFlags2 readStage = readAccess != AccessFlags2.None || writeAccess == AccessFlags2.None + ? target.Stage + : PipelineStageFlags2.None; + bool writes = writeAccess != AccessFlags2.None; + + bool needed; + if (state.Layout != target.Layout) + { + needed = true; + } + else if (writes) + { + // Separate uses can overlap even when they execute at the same stage. + needed = state.ReadStages != PipelineStageFlags2.None || + state.WriteStage != PipelineStageFlags2.None; + } + else + { + // RAW: the last write is not yet visible to this reader's stage. + needed = state.WriteStage != PipelineStageFlags2.None && + (target.Stage & ~state.VisibleStages) != PipelineStageFlags2.None; + } + + sides = default; + if (needed) + { + PipelineStageFlags2 srcStage = state.WriteStage | state.ReadStages; + AccessFlags2 srcAccess = state.WriteAccess | state.ReadAccess; + if (srcStage == PipelineStageFlags2.None) + { + // Nothing used it since the last barrier: that barrier is the prior use. + srcStage = state.VisibleStages; + srcAccess = AccessFlags2.None; + } + + ImageLayout oldLayout = state.Layout; + if (discard && state.Layout != target.Layout) oldLayout = ImageLayout.Undefined; + sides = new BarrierSides(oldLayout, target.Layout, srcStage, srcAccess, target.Stage, target.Access); + // Retain the last producer until another write replaces it: a later + // reader at a different stage still needs that write made visible. + PipelineStageFlags2 visible = state.Layout == target.Layout + ? state.VisibleStages | target.Stage : target.Stage; + state = new SubresourceState(target.Layout, state.WriteStage, state.WriteAccess, + visible, PipelineStageFlags2.None, AccessFlags2.None); + } + + if (writes) + { + state = new SubresourceState(target.Layout, writeStage, writeAccess, + PipelineStageFlags2.None, readStage, readAccess); + } + else + { + state = state with + { + ReadStages = state.ReadStages | readStage, + ReadAccess = state.ReadAccess | readAccess, + VisibleStages = state.VisibleStages | target.Stage, + }; + } + return needed; + } +} + +/// +/// What a command does with an image. Layout, pipeline stage and access all +/// derive from this (), never from the layout +/// alone: two uses can share a layout and still differ in stage (a depth +/// attachment read only by the depth test versus one also sampled by the +/// fragment shader). +/// +public enum ResourceUsage +{ + /// Colour attachment, written without reading the destination. + ColorWrite, + /// Colour attachment with blending: the destination is read and written. + ColorBlend, + /// Depth attachment with writes on. + DepthWrite, + /// Depth attachment with writes off, read by the depth test only. + DepthReadOnly, + /// Depth attachment with writes off, also sampled by the fragment shader. + DepthReadOnlySampled, + /// Sampled by a fragment shader. + SampleFragment, + /// Sampled by a vertex shader. + SampleVertex, + /// Read as a storage image. + StorageRead, + /// Sampled by a compute shader. + SampleCompute, + /// Read as a storage image by a compute shader, never written. + StorageReadCompute, + /// Written as a storage image by a compute shader without reading it first. + StorageWrite, + /// Read and written as a storage image by a compute shader. + StorageReadWrite, + /// Source of a copy or blit. + TransferSrc, + /// Destination of a copy, blit or clear. + TransferDst, + /// Handed to vkQueuePresentKHR. + PresentSrc, +} + +/// +/// The layout, stage and access of one : the +/// destination side of a barrier into that usage. +/// +internal readonly record struct UsageState(ImageLayout Layout, PipelineStageFlags2 Stage, AccessFlags2 Access) +{ + /// The fragment test stages a depth attachment is used at. + public const PipelineStageFlags2 DepthTests = + PipelineStageFlags2.EarlyFragmentTestsBit | PipelineStageFlags2.LateFragmentTestsBit; + + /// Every access bit that writes. + public const AccessFlags2 WriteAccessMask = + AccessFlags2.ColorAttachmentWriteBit | AccessFlags2.DepthStencilAttachmentWriteBit | + AccessFlags2.TransferWriteBit | AccessFlags2.ShaderStorageWriteBit | AccessFlags2.MemoryWriteBit; + + /// + /// The usage table. is the image's aspect: an + /// attachment usage on a depth image resolves to its depth form and a depth + /// attachment usage on a colour image to its colour form, so a caller that + /// only knows "attachment" gets the right one. Sampling keeps + /// SHADER_READ_ONLY_OPTIMAL for both aspects, because that is the layout the + /// descriptor writes name. + /// + // SHADER_READ includes sampled reads. Keep the aggregate access bit for image + // sampling: UHD 770 / Windows driver 101.7088 returns stale texels after + // attachment reuse with SHADER_SAMPLED_READ alone. The aggregate mask fixes + // both the transient post-chain and TAA output reproductions without adding + // a global barrier or widening the pipeline stages. + public static UsageState For(ResourceUsage usage, bool depth) => Normalise(usage, depth) switch + { + ResourceUsage.ColorWrite => new(ImageLayout.ColorAttachmentOptimal, + PipelineStageFlags2.ColorAttachmentOutputBit, AccessFlags2.ColorAttachmentWriteBit), + ResourceUsage.ColorBlend => new(ImageLayout.ColorAttachmentOptimal, + PipelineStageFlags2.ColorAttachmentOutputBit, + AccessFlags2.ColorAttachmentReadBit | AccessFlags2.ColorAttachmentWriteBit), + ResourceUsage.DepthWrite => new(ImageLayout.DepthAttachmentOptimal, DepthTests, + AccessFlags2.DepthStencilAttachmentReadBit | AccessFlags2.DepthStencilAttachmentWriteBit), + ResourceUsage.DepthReadOnly => new(ImageLayout.DepthReadOnlyOptimal, DepthTests, + AccessFlags2.DepthStencilAttachmentReadBit), + ResourceUsage.DepthReadOnlySampled => new(ImageLayout.DepthReadOnlyOptimal, + DepthTests | PipelineStageFlags2.FragmentShaderBit, + AccessFlags2.DepthStencilAttachmentReadBit | AccessFlags2.ShaderReadBit), + ResourceUsage.SampleFragment => new(ImageLayout.ShaderReadOnlyOptimal, + PipelineStageFlags2.FragmentShaderBit, AccessFlags2.ShaderReadBit), + ResourceUsage.SampleVertex => new(ImageLayout.ShaderReadOnlyOptimal, + PipelineStageFlags2.VertexShaderBit, AccessFlags2.ShaderReadBit), + ResourceUsage.StorageRead => new(ImageLayout.General, + PipelineStageFlags2.FragmentShaderBit | PipelineStageFlags2.ComputeShaderBit, + AccessFlags2.ShaderStorageReadBit), + ResourceUsage.SampleCompute => new(ImageLayout.ShaderReadOnlyOptimal, + PipelineStageFlags2.ComputeShaderBit, AccessFlags2.ShaderReadBit), + ResourceUsage.StorageReadCompute => new(ImageLayout.General, + PipelineStageFlags2.ComputeShaderBit, AccessFlags2.ShaderStorageReadBit), + ResourceUsage.StorageWrite => new(ImageLayout.General, + PipelineStageFlags2.ComputeShaderBit, AccessFlags2.ShaderStorageWriteBit), + ResourceUsage.StorageReadWrite => new(ImageLayout.General, + PipelineStageFlags2.ComputeShaderBit, AccessFlags2.ShaderStorageReadBit | AccessFlags2.ShaderStorageWriteBit), + ResourceUsage.TransferSrc => new(ImageLayout.TransferSrcOptimal, + PipelineStageFlags2.TransferBit, AccessFlags2.TransferReadBit), + ResourceUsage.TransferDst => new(ImageLayout.TransferDstOptimal, + PipelineStageFlags2.TransferBit, AccessFlags2.TransferWriteBit), + ResourceUsage.PresentSrc => new(ImageLayout.PresentSrcKhr, + PipelineStageFlags2.BottomOfPipeBit, AccessFlags2.None), + _ => throw new System.ArgumentOutOfRangeException(nameof(usage), usage, null), + }; + + /// + /// The write a usage performs, which the next barrier must make available. + /// An attachment is written by its store op even with writes off: a + /// read-only depth attachment is still stored, and synchronization + /// validation reports the next transition as write-after-write unless the + /// barrier names that write (2026-09-11). + /// + public static (PipelineStageFlags2 Stage, AccessFlags2 Access) WriteOf(ResourceUsage usage, bool depth) => + Normalise(usage, depth) switch + { + ResourceUsage.ColorWrite or ResourceUsage.ColorBlend => + (PipelineStageFlags2.ColorAttachmentOutputBit, AccessFlags2.ColorAttachmentWriteBit), + ResourceUsage.DepthWrite or ResourceUsage.DepthReadOnly or ResourceUsage.DepthReadOnlySampled => + (DepthTests, AccessFlags2.DepthStencilAttachmentWriteBit), + ResourceUsage.TransferDst => (PipelineStageFlags2.TransferBit, AccessFlags2.TransferWriteBit), + // A dispatch's storage write: the next barrier must make it available. + ResourceUsage.StorageWrite or ResourceUsage.StorageReadWrite => + (PipelineStageFlags2.ComputeShaderBit, AccessFlags2.ShaderStorageWriteBit), + _ => (PipelineStageFlags2.None, AccessFlags2.None), + }; + + /// + /// The usage a layout stands for, for callers that still speak in layouts + /// (tests, a readback restoring what it found). Attachment layouts map to + /// the widest use of that layout: blending for colour, sampled for + /// read-only depth. + /// + public static ResourceUsage ForLayout(ImageLayout layout) => layout switch + { + ImageLayout.ShaderReadOnlyOptimal => ResourceUsage.SampleFragment, + ImageLayout.ColorAttachmentOptimal => ResourceUsage.ColorBlend, + ImageLayout.DepthAttachmentOptimal or ImageLayout.DepthStencilAttachmentOptimal => ResourceUsage.DepthWrite, + ImageLayout.DepthReadOnlyOptimal or ImageLayout.DepthStencilReadOnlyOptimal => ResourceUsage.DepthReadOnlySampled, + ImageLayout.TransferSrcOptimal => ResourceUsage.TransferSrc, + ImageLayout.TransferDstOptimal => ResourceUsage.TransferDst, + ImageLayout.PresentSrcKhr => ResourceUsage.PresentSrc, + ImageLayout.General => ResourceUsage.StorageRead, + _ => throw new System.ArgumentOutOfRangeException(nameof(layout), layout, "no usage stands for this layout"), + }; + + private static ResourceUsage Normalise(ResourceUsage usage, bool depth) => (usage, depth) switch + { + (ResourceUsage.ColorWrite, true) => ResourceUsage.DepthWrite, + (ResourceUsage.ColorBlend, true) => ResourceUsage.DepthWrite, + (ResourceUsage.DepthWrite, false) => ResourceUsage.ColorBlend, + (ResourceUsage.DepthReadOnly, false) => ResourceUsage.ColorBlend, + (ResourceUsage.DepthReadOnlySampled, false) => ResourceUsage.ColorBlend, + _ => usage, + }; +} + +/// +/// The readers a buffer can have, derived from its usage flags. Buffers have no +/// layout, so the barriers around a staged copy name every use the buffer was +/// created for instead of ALL_COMMANDS. +/// +internal static class BufferUsageState +{ + public static (PipelineStageFlags2 Stage, AccessFlags2 Access) UsesOf(BufferUsageFlags usage) + { + PipelineStageFlags2 stage = PipelineStageFlags2.None; + AccessFlags2 access = AccessFlags2.None; + if ((usage & BufferUsageFlags.VertexBufferBit) != 0) + { + stage |= PipelineStageFlags2.VertexAttributeInputBit; + access |= AccessFlags2.VertexAttributeReadBit; + } + if ((usage & BufferUsageFlags.IndexBufferBit) != 0) + { + stage |= PipelineStageFlags2.IndexInputBit; + access |= AccessFlags2.IndexReadBit; + } + if ((usage & BufferUsageFlags.UniformBufferBit) != 0) + { + stage |= PipelineStageFlags2.VertexShaderBit | PipelineStageFlags2.FragmentShaderBit; + access |= AccessFlags2.UniformReadBit; + } + if ((usage & BufferUsageFlags.StorageBufferBit) != 0) + { + stage |= PipelineStageFlags2.VertexShaderBit | PipelineStageFlags2.FragmentShaderBit | + PipelineStageFlags2.ComputeShaderBit; + access |= AccessFlags2.ShaderStorageReadBit; + } + if ((usage & BufferUsageFlags.IndirectBufferBit) != 0) + { + stage |= PipelineStageFlags2.DrawIndirectBit; + access |= AccessFlags2.IndirectCommandReadBit; + } + if ((usage & BufferUsageFlags.TransferSrcBit) != 0) + { + stage |= PipelineStageFlags2.TransferBit; + access |= AccessFlags2.TransferReadBit; + } + if ((usage & BufferUsageFlags.TransferDstBit) != 0) + { + // A previous staged copy wrote it: that write must be made available too. + stage |= PipelineStageFlags2.TransferBit; + access |= AccessFlags2.TransferWriteBit; + } + return (stage, access); + } +} diff --git a/Optimum.Render.Vulkan/Graph/TransientAllocator.cs b/Optimum.Render.Vulkan/Graph/TransientAllocator.cs new file mode 100644 index 00000000..a4670bf7 --- /dev/null +++ b/Optimum.Render.Vulkan/Graph/TransientAllocator.cs @@ -0,0 +1,366 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; +using Optimum.Render.Vulkan.Core; + +namespace Optimum.Render.Vulkan.Graph; + +/// What a transient image has to be: two leases with equal descriptions may share one image. +internal readonly record struct TransientImageDesc(uint Width, uint Height, Format Format, uint MipLevels = 1, uint Layers = 1); + +/// +/// One transient lifetime served by a physical image for the current frame. +/// +/// The physical image (a texture id in the device's table). +/// The physical image slot for this frame. +/// First pass of the lifetime, inclusive. +/// Last pass of the lifetime, inclusive. +/// An earlier lease of this frame already used the same image. +/// The image's memory size. +internal readonly record struct TransientLease(int TextureId, int Slot, int FirstPass, int LastPass, bool Aliased, ulong Bytes); + +/// +/// What needs from the device. Separates pooling and lifetime decisions +/// from Vulkan image creation, rebinding and retirement. +/// +internal interface ITransientBacking +{ + /// Creates an image in the Transient memory pool class and returns its texture id. + int Create(TransientImageDesc desc); + + /// Releases an image; the device retires it on the timeline. + void Destroy(int textureId); + + ulong BytesOf(int textureId); + + /// The description of a texture that can be served by a transient image. + bool TryDescribe(int textureId, out TransientImageDesc desc); + + /// The image's contents stop mattering: its next use transitions from UNDEFINED. + void Discard(int textureId); + + /// Makes resolve to the physical image until . + void Rebind(int logicalTextureId, int physicalTextureId); + + /// Undoes every . + void RestoreBindings(); +} + +/// +/// Physical backing for frame-graph transients (Phase 2 step 4). +/// +/// The graph acquires a transient in frame order with its pass lifetime +/// () and gets the physical image that serves it this frame. Images +/// come from the Transient memory pool class and are kept across frames, one pool per +/// : the k-th slot of a description in a frame takes that +/// description's k-th image, so a frame shaped like the last one creates nothing. +/// +/// Aliasing off (the default): every lease gets its own image and keeps its +/// contents like any texture. Aliasing on (OPTIMUM_VULKAN_ALIAS=1): leases +/// use the first compatible slot whose previous lease has ended. Lifetimes that do not +/// overlap share an image, and every lease discards: its first use this frame transitions +/// from UNDEFINED (the previous lease's uses stay on that barrier's source side). +/// +/// Leases arrive with non-decreasing first passes. Placement is incremental: a later +/// lease never moves an earlier one, and equal pass endpoints count as overlapping. +/// +/// Framebuffer slots 2, 3, 4, 7, 8, 9, 10, 13, 14, 15, 18 and 21 (the post chain) opt in: +/// their colour textures are created in the Transient pool and registered +/// (), and the graph serves them through . +/// +internal sealed class TransientAllocator +{ + public const string AliasVariable = "OPTIMUM_VULKAN_ALIAS"; + + /// Frames a pooled image may go unused before it is released. + public const int IdleFrames = 120; + + /// The client framebuffer slots whose colour textures are transient. + public static readonly int[] PostChainSlots = { 2, 3, 4, 7, 8, 9, 10, 13, 14, 15, 18, 21 }; + + public static bool IsPostChainSlot(int slot) => Array.IndexOf(PostChainSlots, slot) >= 0; + + /// Aliasing is off unless OPTIMUM_VULKAN_ALIAS=1. + public static bool AliasingFromEnvironment() => Environment.GetEnvironmentVariable(AliasVariable) == "1"; + + private sealed class PhysicalImage + { + public int TextureId; + public TransientImageDesc Description; + public int LastPass; + public ulong Bytes; + public long LastUsedFrame; + } + + private readonly ITransientBacking _backing; + private readonly Dictionary> _pools = new(); + private readonly Dictionary _usedThisFrame = new(); + private readonly List _slotImages = new(); + private readonly List _leases = new(); + private readonly Dictionary _optedIn = new(); + private bool _aliasing; + private long _frame; + private int _lastFirstPass = -1; + + public TransientAllocator(ITransientBacking backing, bool aliasing) + { + _backing = backing ?? throw new ArgumentNullException(nameof(backing)); + _aliasing = aliasing; + } + + /// Whether leases share images. Changes only between frames. + public bool Aliasing + { + get => _aliasing; + set + { + if (_leases.Count > 0) throw new InvalidOperationException("aliasing changes between frames only"); + _aliasing = value; + } + } + + /// The leases handed out since , in order. + public IReadOnlyList Leases => _leases; + + /// Physical images held across frames. + public int PhysicalImageCount + { + get + { + int count = 0; + foreach (List pool in _pools.Values) count += pool.Count; + return count; + } + } + + /// Bytes of the physical images held. + public ulong PhysicalBytes + { + get + { + ulong bytes = 0; + foreach (List pool in _pools.Values) + { + foreach (PhysicalImage image in pool) bytes += image.Bytes; + } + return bytes; + } + } + + /// Bytes of the opted-in logical textures (their own images). + public ulong OptedInBytes + { + get + { + ulong bytes = 0; + foreach (int id in _optedIn.Keys) bytes += _backing.BytesOf(id); + return bytes; + } + } + + /// Bytes of this frame's leases served by an image an earlier lease already used. + public ulong AliasedBytes + { + get + { + ulong bytes = 0; + foreach (TransientLease lease in _leases) + { + if (lease.Aliased) bytes += lease.Bytes; + } + return bytes; + } + } + + public int AliasedLeaseCount + { + get + { + int count = 0; + foreach (TransientLease lease in _leases) + { + if (lease.Aliased) count++; + } + return count; + } + } + + /// Registers a client texture as a transient resource of framebuffer . + public void OptIn(int logicalTextureId, int framebufferSlot) + { + if (logicalTextureId <= 0) return; + _optedIn[logicalTextureId] = framebufferSlot; + } + + /// Drops a deleted texture from the opt-in set. + public void Forget(int logicalTextureId) => _optedIn.Remove(logicalTextureId); + + public bool IsOptedIn(int textureId) => _optedIn.ContainsKey(textureId); + + /// The framebuffer slot an opted-in texture belongs to, or -1. + public int SlotOf(int textureId) => _optedIn.TryGetValue(textureId, out int slot) ? slot : -1; + + public int OptedInCount => _optedIn.Count; + + /// + /// Ends the previous frame's leases and bindings, and releases images that went + /// unused for frames. Call once per frame, before the + /// first . + /// + public void BeginFrame() + { + _backing.RestoreBindings(); + _slotImages.Clear(); + _usedThisFrame.Clear(); + _leases.Clear(); + _lastFirstPass = -1; + _frame++; + Trim(); + } + + /// + /// The physical image serving a transient with pass lifetime + /// [, ] this frame. + /// Leases arrive in frame order: a first pass below an earlier lease's is rejected. + /// + public TransientLease Acquire(TransientImageDesc desc, int firstPass, int lastPass) + { + if (firstPass < 0) throw new ArgumentOutOfRangeException(nameof(firstPass), firstPass, "negative pass"); + if (lastPass < firstPass) throw new ArgumentOutOfRangeException(nameof(lastPass), lastPass, "ends before it starts"); + if (firstPass < _lastFirstPass) + { + throw new InvalidOperationException( + "transients are acquired in frame order: first pass " + firstPass + " after " + _lastFirstPass); + } + _lastFirstPass = firstPass; + + int slot = _slotImages.Count; + if (_aliasing) + { + for (int i = 0; i < _slotImages.Count; i++) + { + PhysicalImage candidate = _slotImages[i]; + if (candidate.Description == desc && candidate.LastPass < firstPass) + { + slot = i; + break; + } + } + } + + bool aliased = slot < _slotImages.Count; + PhysicalImage image; + if (aliased) + { + image = _slotImages[slot]; + } + else + { + image = Take(desc); + _slotImages.Add(image); + } + + image.LastPass = lastPass; + image.LastUsedFrame = _frame; + if (_aliasing) _backing.Discard(image.TextureId); + + var lease = new TransientLease(image.TextureId, slot, firstPass, lastPass, aliased, image.Bytes); + _leases.Add(lease); + return lease; + } + + /// + /// Serves a client texture for [, ] + /// this frame and returns the texture id that now backs it. With aliasing off the texture + /// keeps its own image; with aliasing on its id resolves to a leased image until the next + /// . + /// + public int Bind(int logicalTextureId, int firstPass, int lastPass) + { + if (!_aliasing) return logicalTextureId; + if (!_backing.TryDescribe(logicalTextureId, out TransientImageDesc desc)) return logicalTextureId; + TransientLease lease = Acquire(desc, firstPass, lastPass); + _backing.Rebind(logicalTextureId, lease.TextureId); + return lease.TextureId; + } + + private PhysicalImage Take(TransientImageDesc desc) + { + if (!_pools.TryGetValue(desc, out List? pool)) + { + pool = new List(); + _pools.Add(desc, pool); + } + + _usedThisFrame.TryGetValue(desc, out int index); + _usedThisFrame[desc] = index + 1; + if (index < pool.Count) return pool[index]; + + int id = _backing.Create(desc); + var image = new PhysicalImage { TextureId = id, Description = desc, Bytes = _backing.BytesOf(id), LastUsedFrame = _frame }; + pool.Add(image); + return image; + } + + /// Pools serve the k-th slot with the k-th image, so only trailing images can go. + private void Trim() + { + foreach (List pool in _pools.Values) + { + while (pool.Count > 0 && _frame - pool[pool.Count - 1].LastUsedFrame > IdleFrames) + { + _backing.Destroy(pool[pool.Count - 1].TextureId); + pool.RemoveAt(pool.Count - 1); + } + } + } +} + +/// +/// over the device's texture table: images in the +/// Transient memory pool class, released through the device (descriptor eviction, then +/// destruction once the timelines passed), rebinding by texture id. +/// +internal sealed class TextureTransientBacking : ITransientBacking +{ + private readonly TextureManager _textures; + private readonly Action _release; + + /// The device's texture table. + /// The device's texture release (evicts descriptor sets, retires on the timeline). + public TextureTransientBacking(TextureManager textures, Action release) + { + _textures = textures; + _release = release; + } + + public int Create(TransientImageDesc desc) => + _textures.Create(desc.Width, desc.Height, desc.Format, layers: desc.Layers, + generateMipmaps: desc.MipLevels > 1, poolClass: MemoryPoolClass.Transient); + + public void Destroy(int textureId) => _release(textureId); + + public ulong BytesOf(int textureId) => _textures.Get(textureId)?.Allocation.Size ?? 0; + + public bool TryDescribe(int textureId, out TransientImageDesc desc) + { + VulkanTexture? texture = _textures.Get(textureId); + if (texture == null || texture.Cube || texture.Aspect != ImageAspectFlags.ColorBit) + { + desc = default; + return false; + } + desc = new TransientImageDesc(texture.Width, texture.Height, texture.Format, texture.MipLevels, texture.Layers); + return true; + } + + public void Discard(int textureId) + { + VulkanTexture? texture = _textures.Get(textureId); + if (texture != null) _textures.DiscardContents(texture); + } + + public void Rebind(int logicalTextureId, int physicalTextureId) => _textures.Rebind(logicalTextureId, physicalTextureId); + + public void RestoreBindings() => _textures.RestoreBindings(); +} diff --git a/Optimum.Render.Vulkan/Latency/FrameTiming.cs b/Optimum.Render.Vulkan/Latency/FrameTiming.cs new file mode 100644 index 00000000..41f31a5d --- /dev/null +++ b/Optimum.Render.Vulkan/Latency/FrameTiming.cs @@ -0,0 +1,269 @@ +using System; +using System.Diagnostics; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// CPU-observed frame intervals in microseconds. +/// TotalUs ends when vkQueuePresentKHR returns, not when the display scans out. +/// +internal readonly record struct LatencyFrameReport( + ulong FrameId, + ulong PresentId, + ulong InputUs, + ulong SimulationUs, + ulong RenderSubmitUs, + ulong PresentUs, + ulong TotalUs) +{ + /// + /// Builds a report from CPU timestamps on one monotonic clock in microseconds (see + /// ). A missing marker is passed as 0 and makes the + /// intervals that need it 0; intervals never go negative. + /// + public static LatencyFrameReport FromCpuTimestamps( + ulong frameId, + ulong presentId, + long inputSampleUs, + long simulationStartUs, + long simulationEndUs, + long renderSubmitStartUs, + long renderSubmitEndUs, + long presentStartUs, + long presentEndUs) + { + long start = inputSampleUs != 0 ? inputSampleUs : simulationStartUs; + return new LatencyFrameReport( + frameId, + presentId, + Span(inputSampleUs, simulationStartUs), + Span(simulationStartUs, simulationEndUs), + Span(renderSubmitStartUs, renderSubmitEndUs), + Span(presentStartUs, presentEndUs), + Span(start, presentEndUs)); + } + + /// 0 when either end is missing or the pair is out of order; the difference otherwise. + private static ulong Span(long fromUs, long toUs) => + fromUs <= 0 || toUs <= 0 || toUs <= fromUs ? 0UL : (ulong)(toUs - fromUs); +} + +/// CPU phase boundaries owned by the input, render and present call sites. +internal enum LatencyMarker +{ + SimulationStart, + SimulationEnd, + RenderSubmitStart, + RenderSubmitEnd, + PresentStart, + PresentEnd, + InputSample, +} + +/// One monotonic microsecond clock for every latency timestamp. +internal static class LatencyClock +{ + private static readonly double TicksToUs = 1_000_000.0 / Stopwatch.Frequency; + + /// Microseconds since an arbitrary origin; never 0, so 0 stays "no timestamp". + public static long NowUs() + { + long us = (long)(Stopwatch.GetTimestamp() * TicksToUs); + return us > 0 ? us : 1; + } +} + +/// +/// Collects the current frame's CPU phases. A new frame discards incomplete +/// timestamps and logs the interruption once, so missing markers cannot leak +/// into another frame or repeatedly fill the client log. +/// +internal sealed class LatencyPhaseTracker +{ + private readonly Action? _log; + + private ulong _frameId; + private bool _frameOpen; + private bool _logged; + + private long _inputSample; + private long _simStart; + private long _simEnd; + private long _renderSubmitStart; + private long _renderSubmitEnd; + private long _presentStart; + private long _presentEnd; + + public LatencyPhaseTracker(Action? log = null) => _log = log; + + /// How often a phase was still open when the next frame started. + public int SelfHealCount { get; private set; } + + /// Whether the self-heal note has already gone to the log. + public bool SelfHealLogged => _logged; + + /// The frame currently collecting markers; 0 before the first one. + public ulong CurrentFrameId => _frameId; + + /// True while some Start has no matching End in the current frame. + public bool HasOpenPhase => _frameOpen && + ((_simStart != 0 && _simEnd == 0) || + (_renderSubmitStart != 0 && _renderSubmitEnd == 0) || + (_presentStart != 0 && _presentEnd == 0)); + + /// + /// Stamps one marker. A marker of a frame id other than the current one + /// starts a new frame first, closing whatever the old one left open. + /// + public void Mark(ulong frameId, LatencyMarker marker, long timestampUs) + { + if (!_frameOpen || frameId != _frameId) BeginFrame(frameId); + + switch (marker) + { + case LatencyMarker.InputSample: _inputSample = timestampUs; break; + case LatencyMarker.SimulationStart: _simStart = timestampUs; break; + case LatencyMarker.SimulationEnd: _simEnd = timestampUs; break; + case LatencyMarker.RenderSubmitStart: _renderSubmitStart = timestampUs; break; + case LatencyMarker.RenderSubmitEnd: _renderSubmitEnd = timestampUs; break; + case LatencyMarker.PresentStart: _presentStart = timestampUs; break; + case LatencyMarker.PresentEnd: _presentEnd = timestampUs; break; + } + } + + /// + /// Starts a frame: discards any phases the previous frame left open (counted, + /// and logged the first time only) and clears the timestamps. + /// + private void BeginFrame(ulong frameId) + { + if (_frameOpen && HasOpenPhase) + { + SelfHealCount++; + if (!_logged) + { + _logged = true; + _log?.Invoke("latency: a phase of frame " + _frameId + + " was still open when frame " + frameId + + " started; closing it. Reported once, however often it happens."); + } + } + + _frameId = frameId; + _frameOpen = true; + _inputSample = 0; + _simStart = 0; + _simEnd = 0; + _renderSubmitStart = 0; + _renderSubmitEnd = 0; + _presentStart = 0; + _presentEnd = 0; + } + + /// + /// Closes the frame and builds its report. False when + /// is not the frame being collected (a present of + /// a frame whose markers never arrived), leaving the state untouched. + /// + public bool TryComplete(ulong frameId, ulong presentId, out LatencyFrameReport report) + { + if (!_frameOpen || frameId != _frameId) + { + report = default; + return false; + } + + report = LatencyFrameReport.FromCpuTimestamps( + frameId, presentId, + _inputSample, _simStart, _simEnd, + _renderSubmitStart, _renderSubmitEnd, + _presentStart, _presentEnd); + _frameOpen = false; + return true; + } +} + +/// +/// Bounded completed-frame reports. Overflow drops the oldest entry in constant +/// time; taking reports returns chronological order and empties the buffer. +/// +internal sealed class LatencyReportBuffer +{ + /// A few seconds of frames; a client that never samples must not grow this. + public const int DefaultCapacity = 256; + + private readonly LatencyFrameReport[] _reports; + private readonly object _lock = new(); + + /// Where the next report is written. + private int _next; + + /// How many of the slots hold a report that has not been taken. + private int _count; + + public LatencyReportBuffer(int capacity = DefaultCapacity) + { + if (capacity < 1) capacity = 1; + _reports = new LatencyFrameReport[capacity]; + } + + public int Capacity => _reports.Length; + + /// How many reports are waiting to be taken. + public int Count + { + get { lock (_lock) return _count; } + } + + /// Adds one report, dropping the oldest when the ring is full. + public void Add(in LatencyFrameReport report) + { + lock (_lock) + { + _reports[_next] = report; + _next = _next + 1 == _reports.Length ? 0 : _next + 1; + if (_count < _reports.Length) _count++; + } + } + + /// The reports since the last call, oldest first, and clears them. + public LatencyFrameReport[] Take() + { + lock (_lock) + { + if (_count == 0) return Array.Empty(); + + var taken = new LatencyFrameReport[_count]; + int index = _next - _count; + if (index < 0) index += _reports.Length; + for (int i = 0; i < _count; i++) + { + taken[i] = _reports[index]; + index = index + 1 == _reports.Length ? 0 : index + 1; + } + + _count = 0; + _next = 0; + return taken; + } + } +} + +/// CPU phase timestamps and bounded reports for completed frames. +internal sealed class FrameTimingRecorder +{ + private readonly LatencyPhaseTracker _tracker; + private readonly LatencyReportBuffer _reports = new(); + + public FrameTimingRecorder(Action? log = null) => _tracker = new LatencyPhaseTracker(log); + + public void Marker(ulong frameId, LatencyMarker marker) => + _tracker.Mark(frameId, marker, LatencyClock.NowUs()); + + public void OnPresent(ulong frameId, ulong presentId) + { + if (_tracker.TryComplete(frameId, presentId, out LatencyFrameReport report)) _reports.Add(report); + } + + public LatencyFrameReport[] TakeReports() => _reports.Take(); +} diff --git a/Optimum.Render.Vulkan/Optimum.Render.Vulkan.csproj b/Optimum.Render.Vulkan/Optimum.Render.Vulkan.csproj new file mode 100644 index 00000000..45360b7c --- /dev/null +++ b/Optimum.Render.Vulkan/Optimum.Render.Vulkan.csproj @@ -0,0 +1,104 @@ + + + + + + net10.0 + Optimum.Render.Vulkan + Optimum.Render.Vulkan + true + ..\bin\$(Configuration) + annotations + + true + true + + + + + + + false + all + + + + + + + + + + + + + + + + + ..\.vanilla\win-x64\vintagestory\VintagestoryAPI.dll + false + + + + ..\.vanilla\win-x64\vintagestory\Lib\cairo-sharp.dll + false + + + ..\.vanilla\win-x64\vintagestory\Lib\OpenTK.Audio.OpenAL.dll + false + + + ..\.vanilla\win-x64\vintagestory\Lib\OpenTK.Windowing.Desktop.dll + false + + + ..\.vanilla\win-x64\vintagestory\Lib\OpenTK.Windowing.Common.dll + false + + + ..\.vanilla\win-x64\vintagestory\Lib\OpenTK.Mathematics.dll + false + + + ..\.vanilla\win-x64\vintagestory\Lib\OpenTK.Graphics.dll + false + + + + + + + + + + + + + shaders-vk/gtao/%(Filename)%(Extension) + + + + diff --git a/Optimum.Render.Vulkan/Platform/StatedRenderState.cs b/Optimum.Render.Vulkan/Platform/StatedRenderState.cs new file mode 100644 index 00000000..3c906e57 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/StatedRenderState.cs @@ -0,0 +1,378 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Optimum.Render.Vulkan.Graph; + +namespace Optimum.Render.Vulkan.Platform; + +/// +/// The fixed-function state the client stated through the platform's virtuals, with OpenGL's +/// semantics, owned by the platform (docs/vulkan.md, decision 3: a native +/// system reads client state, never the device's). The generic native draw +/// (VulkanClientPlatform.NativeStated.cs) builds its pipeline, pass and textures from this alone. +/// +/// Semantics that differ from a naive record, each as the OpenGL body does it: +/// - blend: glBlendFunc sets every draw buffer, glBlendFunci one; glDisable(GL_BLEND) +/// keeps the functions (, , ); +/// - colour mask: glColorMask is global (); +/// - draw buffers: per framebuffer, and a framebuffer nobody selected for writes attachment 0 only, +/// GL's default for a framebuffer object (); +/// - texture units: one texture per unit, whatever its dimensionality, as the device's table has it +/// (a cube and a 2D bind to the same unit replace each other there too); +/// - viewport: a framebuffer bind does not change it; the platform's own bind states the full target. +/// Stencil is recorded but never applied: no framebuffer of this client has a stencil attachment, +/// so a stencil test passes on either path. +/// +internal sealed class StatedRenderState +{ + public const int MaxColorAttachments = RenderLimits.MaxColorAttachments; + public const int MaxTextureUnits = RenderLimits.MaxTextureUnits; + + private readonly AttachmentBlend[] _blend = new AttachmentBlend[MaxColorAttachments]; + private readonly Dictionary<(int FramebufferId, int Count), AttachmentBlend[]> _blendSnapshots = new(); + private readonly Dictionary _drawBuffers = new(); + private readonly int[] _unitTextures = new int[MaxTextureUnits]; + private readonly int[] _unitSamplers = new int[MaxTextureUnits]; + + public StatedRenderState() + { + for (int i = 0; i < _blend.Length; i++) _blend[i] = AttachmentBlend.Default; + } + + public bool BlendEnabled { get; private set; } + + /// The channels glColorMask left writable, applied to every attachment. + public ColorComponentFlags ColorMask { get; private set; } = + ColorComponentFlags.RBit | ColorComponentFlags.GBit | ColorComponentFlags.BBit | ColorComponentFlags.ABit; + + public bool DepthTest { get; set; } + public bool DepthWrite { get; set; } = true; + public CompareOp DepthCompare { get; set; } = CompareOp.Less; + public bool CullEnabled { get; set; } + public bool CullBack { get; set; } = true; + public float LineWidth { get; set; } = 1f; + public bool Wireframe { get; set; } + public bool ScissorEnabled { get; set; } + public Rect2D Scissor { get; set; } + public Rect2D Viewport { get; set; } + public bool StencilTest { get; set; } + + public CullModeFlags CullMode => CullEnabled ? (CullBack ? CullModeFlags.BackBit : CullModeFlags.FrontBit) : CullModeFlags.None; + + /// glBlendFunc/glBlendFuncSeparate of a named mode: every attachment's factors, add equations. + public void SetBlendMode(EnumBlendMode mode) + { + _blendSnapshots.Clear(); + (BlendFactor srcColor, BlendFactor dstColor, BlendFactor srcAlpha, BlendFactor dstAlpha) = AttachmentBlend.FactorsFor(mode); + for (int i = 0; i < _blend.Length; i++) + { + _blend[i].SrcColor = srcColor; + _blend[i].DstColor = dstColor; + _blend[i].SrcAlpha = srcAlpha; + _blend[i].DstAlpha = dstAlpha; + _blend[i].ColorOp = BlendOp.Add; + _blend[i].AlphaOp = BlendOp.Add; + } + } + + /// glEnable/glDisable(GL_BLEND): the functions stay. + public void SetBlendEnabled(bool enabled) + { + if (BlendEnabled == enabled) return; + BlendEnabled = enabled; + _blendSnapshots.Clear(); + } + + /// glBlendEquationi + glBlendFuncSeparatei with GL tokens. + public void SetSlotBlend(int slot, int glEquation, int srcColor, int dstColor, int srcAlpha, int dstAlpha) + { + if ((uint)slot >= MaxColorAttachments) return; + _blendSnapshots.Clear(); + BlendOp op = GlEnums.BlendOpFrom(glEquation); + _blend[slot].ColorOp = op; + _blend[slot].AlphaOp = op; + _blend[slot].SrcColor = GlEnums.BlendFactorFrom(srcColor); + _blend[slot].DstColor = GlEnums.BlendFactorFrom(dstColor); + _blend[slot].SrcAlpha = GlEnums.BlendFactorFrom(srcAlpha); + _blend[slot].DstAlpha = GlEnums.BlendFactorFrom(dstAlpha); + } + + /// glBlendEquationi: one attachment's equation, its factors kept. + public void SetSlotEquation(int slot, int glEquation) + { + if ((uint)slot >= MaxColorAttachments) return; + _blendSnapshots.Clear(); + BlendOp op = GlEnums.BlendOpFrom(glEquation); + _blend[slot].ColorOp = op; + _blend[slot].AlphaOp = op; + } + + /// glBlendFuncSeparatei: one attachment's factors, its equation kept. + public void SetSlotFunc(int slot, int srcColor, int dstColor, int srcAlpha, int dstAlpha) + { + if ((uint)slot >= MaxColorAttachments) return; + _blendSnapshots.Clear(); + _blend[slot].SrcColor = GlEnums.BlendFactorFrom(srcColor); + _blend[slot].DstColor = GlEnums.BlendFactorFrom(dstColor); + _blend[slot].SrcAlpha = GlEnums.BlendFactorFrom(srcAlpha); + _blend[slot].DstAlpha = GlEnums.BlendFactorFrom(dstAlpha); + } + + public void SetColorMask(bool r, bool g, bool b, bool a) + { + ColorComponentFlags next = (r ? ColorComponentFlags.RBit : 0) | (g ? ColorComponentFlags.GBit : 0) | + (b ? ColorComponentFlags.BBit : 0) | (a ? ColorComponentFlags.ABit : 0); + if (ColorMask == next) return; + ColorMask = next; + _blendSnapshots.Clear(); + } + + /// + /// One attachment as a draw into applies it: the stated + /// functions and enable, written only when the attachment is a selected draw buffer and + /// the colour mask allows it. + /// + public AttachmentBlend AttachmentFor(int framebufferId, int slot) + { + if ((uint)slot >= MaxColorAttachments) return new AttachmentBlend { WriteMask = 0 }; + AttachmentBlend blend = _blend[slot]; + blend.Enabled = BlendEnabled; + blend.WriteMask = ((DrawBuffers(framebufferId) >> slot) & 1) != 0 ? ColorMask : 0; + if (IsUiImage(framebufferId)) blend = blend.ForUiImage(); + return blend; + } + + /// + /// An immutable snapshot for a pipeline description. Pipeline entries retain + /// the array, so invalidation drops this cache's reference without changing + /// descriptions already recorded for earlier draws. + /// + public AttachmentBlend[] BlendFor(int framebufferId, int count) + { + var key = (framebufferId, count); + if (_blendSnapshots.TryGetValue(key, out AttachmentBlend[]? snapshot)) + return snapshot; + + snapshot = new AttachmentBlend[count]; + for (int i = 0; i < count; i++) snapshot[i] = AttachmentFor(framebufferId, i); + _blendSnapshots.Add(key, snapshot); + return snapshot; + } + + /// + /// World/UI separation: the UI image's target while the platform's UI scope is open, 0 otherwise + /// (VulkanClientPlatform.UiSeparation.cs). While it is set, Default means that image. + /// + private int _uiImageFramebuffer; + public int UiImageFramebuffer + { + get => _uiImageFramebuffer; + set + { + if (_uiImageFramebuffer == value) return; + _uiImageFramebuffer = value; + _blendSnapshots.Clear(); + } + } + + /// Whether a draw into lands in the UI image. + public bool IsUiImage(int framebufferId) => + UiImageFramebuffer > 0 && + (framebufferId == Graph.PassDeclaration.DefaultFramebuffer || framebufferId == UiImageFramebuffer); + + public void SetDrawBuffers(int framebufferId, uint mask) + { + if (_drawBuffers.TryGetValue(framebufferId, out uint current) && current == mask) return; + _drawBuffers[framebufferId] = mask; + _blendSnapshots.Clear(); + } + + public uint DrawBuffers(int framebufferId) => + _drawBuffers.TryGetValue(framebufferId, out uint mask) ? mask : 1u; + + public void ForgetFramebuffer(int framebufferId) + { + _drawBuffers.Remove(framebufferId); + _blendSnapshots.Clear(); + } + + public void BindTexture(int unit, int textureId) + { + if ((uint)unit < MaxTextureUnits) _unitTextures[unit] = textureId; + } + + public int TextureAt(int unit) => (uint)unit < MaxTextureUnits ? _unitTextures[unit] : 0; + + public void BindSampler(int unit, int samplerId) + { + if ((uint)unit < MaxTextureUnits) _unitSamplers[unit] = samplerId; + } + + public int SamplerAt(int unit) => (uint)unit < MaxTextureUnits ? _unitSamplers[unit] : 0; +} + +/// +/// One draw of a program recorded natively from a into an +/// explicit target: the generic native draw (VulkanClientPlatform.NativeStated.cs) and the GPU +/// tests' GL-shaped helpers both record through here, so the tests exercise the route the client's +/// unrecognised draws take. +/// +/// What it states, and from where: +/// - target: ; every colour slot attached to it on the device is in +/// the pass (not the FrameBufferRef's own list: the OIT accumulation targets are attached to +/// Transparent at slots 3-5 without being in its ColorTextureIds, and a pass without them drops the +/// accumulated colour - 2026-09-17, water drew black until the slots came from the attachments); +/// the draw buffers stated for that target become per-attachment write masks (decision 4); +/// - blend, colour mask, depth, cull, line width, polygon mode, viewport and scissor: the stated state; +/// - textures: per sampler, the texture on the unit the program points it at (its SetSamplerUnit +/// mapping, else the sampler's declaration order), with the unit's standalone sampler if one is bound; +/// - the depth attachment of the target sampled with depth writes off is read in the read-only +/// layout (SamplesBoundDepth); sampled while written, the draw is refused. +/// +internal static class StatedDraw +{ + /// + /// Records the draw. 0 is the fullscreen triangle; + /// is a pool's multi-draw. False with a reason: nothing was recorded. + /// names the pass the draw belongs to (its name, slots, reads and + /// flags); without one the draw opens "Stated/<target>" over every attached slot. + /// + internal static bool Record(VulkanDevice device, StatedRenderState stated, int programId, int framebufferId, + int meshId, int instances, int[]? starts, int[]? sizes, int groupCount, out string? refusal, + PassDeclaration? declared = null) + { + refusal = null; + RenderTargetFormats? all = device.NativeTargetFormats(framebufferId, uint.MaxValue); + if (all == null) return Refused("framebuffer " + framebufferId + " does not exist", out refusal); + int attached = all.ColorFormats.Length; + uint slots = attached >= 32 ? uint.MaxValue : (1u << attached) - 1u; + if (declared != null) slots &= declared.ColorSlots; + + int layoutId = meshId > 0 ? device.NativeMeshLayoutId(meshId) : MeshManager.EmptyLayoutId; + if (layoutId < 0) return Refused("the mesh has no layout", out refusal); + + // Every sampler the program declares, from the unit it points at. + string[] names = device.SamplerNamesOf(programId); + int[] samplerUnits = device.NativeSamplerUnitsOf(programId); + int depthTexture = device.NativeFramebufferDepthTexture(framebufferId); + bool samplesBoundDepth = false; + var reads = new int[names.Length]; + Span units = stackalloc int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + units[i] = i < samplerUnits.Length ? samplerUnits[i] : -1; + reads[i] = stated.TextureAt(units[i]); + if (reads[i] != 0 && reads[i] == depthTexture) + { + if (stated.DepthWrite && stated.DepthTest) + { + return Refused("it samples the depth attachment it writes", out refusal); + } + samplesBoundDepth = true; + } + } + + // A colour slot the draw samples while its draw buffer is off leaves the pass: GL reads it as + // any texture (the composition writes Primary 0 and reads Primary 1). With its draw buffer on + // it stays, and the device samples a copy of it (feedback). + uint drawBuffers = stated.DrawBuffers(framebufferId); + for (int slot = 0; slot < attached && slot < 32; slot++) + { + if (((drawBuffers >> slot) & 1) != 0 || ((slots >> slot) & 1) == 0) continue; + int attachment = device.NativeFramebufferColorTexture(framebufferId, slot); + if (attachment != 0 && Array.IndexOf(reads, attachment) >= 0) slots &= ~(1u << slot); + } + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, slots); + if (formats == null) return Refused("no formats for framebuffer " + framebufferId, out refusal); + + AttachmentBlend[] blend = stated.BlendFor(framebufferId, Math.Max(formats.ColorFormats.Length, 1)); + + var description = new NativePipelineDescription + { + ProgramId = programId, + Blend = blend, + DepthTest = stated.DepthTest, + DepthWrite = stated.DepthWrite && !samplesBoundDepth, + DepthCompare = stated.DepthCompare, + Cull = stated.CullMode, + Topology = meshId > 0 ? device.NativeMeshTopology(meshId) : PrimitiveTopology.TriangleList, + PolygonMode = stated.Wireframe ? PolygonMode.Line : PolygonMode.Fill, + LineWidth = stated.LineWidth, + VertexLayoutId = layoutId, + SamplesBoundDepth = samplesBoundDepth, + Targets = formats, + }; + NativePipeline? pipeline = device.RequestNativePipeline(description, out string error); + if (pipeline == null) return Refused(error, out refusal); + + var textures = new NativeTexture[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int sampler = stated.SamplerAt(units[i]); + textures[i] = new NativeTexture(pipeline.SamplerAt(i), reads[i], + sampler != 0 ? device.NativeStandaloneSampler(sampler) : null); + } + + int[] passReads = reads; + if (declared != null && declared.Reads.Length > 0) + passReads = MergeReads(declared.Reads, reads); + Rect2D viewport = stated.Viewport; + bool drawn = false; + if (device.BeginNativePass(new NativePassDescription + { + Name = declared?.Name ?? "Stated/" + framebufferId, + FramebufferId = framebufferId, + ColorSlots = slots, + Reads = passReads, + TransientSlots = declared?.TransientSlots ?? 0, + Flags = declared?.Flags ?? PassFlags.AllowSplit, + Generic = true, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + Scissor = stated.ScissorEnabled ? stated.Scissor : null, + })) + { + drawn = meshId <= 0 + ? device.DrawNativeFullscreen(pipeline, textures) + : starts != null + ? device.DrawNativeMeshMulti(pipeline, meshId, starts, sizes!, groupCount, textures) + : device.DrawNativeMeshInstanced(pipeline, meshId, instances, textures); + } + // The scope stays open: the next stated draw on the same target and slots coalesces into + // this pass instead of ending the rendering scope and starting another; anything else + // declares its own pass, which ends this one. + device.EndNativePass(keepScope: true); + return drawn; + } + + private static bool Refused(string reason, out string? refusal) + { + refusal = reason; + return false; + } + + private static int[] MergeReads(int[] declared, int[] sampled) + { + int extra = 0; + for (int i = 0; i < sampled.Length; i++) + { + int read = sampled[i]; + if (Array.IndexOf(declared, read) < 0 && Array.IndexOf(sampled, read, 0, i) < 0) + extra++; + } + + var merged = new int[declared.Length + extra]; + Array.Copy(declared, merged, declared.Length); + int next = declared.Length; + foreach (int read in sampled) + { + if (Array.IndexOf(merged, read, 0, next) < 0) + merged[next++] = read; + } + return merged; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Frame.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Frame.cs new file mode 100644 index 00000000..5dd61d8f --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Frame.cs @@ -0,0 +1,187 @@ +using Optimum.Render.Vulkan.Core; +using Vintagestory.API.Config; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: the frame bracket, the thick-line capability, the +// window-size notification and the parity-dump readback - the device calls that used to +// sit in ClientPlatformWindows.window_RenderFrame, Start, Window_Resize and +// OptimumParityDumpAttachment. The base keeps the frame pacing, the frame handler call and +// the parity dump itself. +public partial class VulkanClientPlatform +{ + /// The client retains ownership of its FPS limiter. + public override bool LatencyOwnsFrameCap => false; + + /// The existing pre-input hook starts frame identity and CPU timing. + public override void LatencySleep() + { + if (device == null) return; + ulong frameId = device.BeginLatencyFrame(); + device.Latency.Marker(frameId, LatencyMarker.InputSample); + device.Latency.Marker(frameId, LatencyMarker.SimulationStart); + } + + /// Recycles the frame slot and opens a command buffer. + public override void BeginFrame() + { + // World/UI separation: a GUI renderer that threw out of the AfterBlit or Ortho stage + // unwound past both compose call sites; the new frame starts with the scope closed, so the + // world pass draws to the targets it names under the factors it states. + CloseUiScope(); + device.BeginFrame(); + // Until a stage or a post method says otherwise, passes are named after the frame. + passContext = "Frame"; + passContextFlags = Graph.PassFlags.AllowSplit; + } + + /// + /// Closes the rendering scope, submits, and blits the result into the swapchain. Runs + /// after the frame handler and the parity dump, exactly where GL swaps buffers. + /// + public override void EndFrame() + { + device.Present(); + } + + /// GL probes by setting a width; the device answers it as a capability. + public override bool ProbeThickLineSupport() + { + // GL leaves the probed width set, so the stated line width is 1.5 from here on too. + stated.LineWidth = 1.5f; + return device.SupportsThickLines; + } + + /// + /// The device owns the default render target and the swapchain, and neither follows + /// the window on its own. The base calls this before RebuildFrameBuffers. + /// + public override void OnWindowSizeChanged(int width, int height) + { + device.Resize(width, height); + } + + /// The single call site of the device's parity readback. + public override OptimumTextureReadback ReadTextureForParity(int textureId) + { + return device.ReadTextureForParity(textureId); + } +} + +/// +/// The first render stage of a frame, as the latency seams see it (seam S4): simulation is +/// over and the renderer starts recording. The bracket fires for every stage; the listener +/// decides which one is the frame's first, keyed on the latency frame id, so no marker of a +/// frame is ever stamped twice. +/// +/// A second listener beside rather +/// than another call inside the frame graph's listener: the two have different lifetimes +/// (the graph listener exists only while a device is installed and is replaced with the +/// graph) and a test drives either on its own. +/// +internal interface ILatencyStageListener +{ + void OnFrameRenderStart(); +} + +// Vulkan-native plan, Phase 2 (contract C3): the render-stage bracket. ClientMain.TriggerRenderStage +// calls BeginRenderStage before the stage's renderers and EndRenderStage after them; the +// platform records the stage and forwards both to the frame graph once it listens. +public partial class VulkanClientPlatform +{ + /// + /// The frame graph's hook; null until the frame graph sets it, in which case the bracket + /// only updates . + /// + internal IRenderStageListener? RenderStageListener; + + /// The stage most recently begun (still valid after it ended). + internal EnumRenderStage CurrentRenderStage { get; private set; } + + /// True between a stage's Begin and End. + internal bool InRenderStage { get; private set; } + + /// + /// The latency hook (seam S4), told at the first stage of each frame. Null means the + /// installed device is used, which is what the client does; a test sets this to watch + /// the bracket without a device. + /// + internal ILatencyStageListener? LatencyStageListener; + + /// + /// The explicit listener, or the installed device, which is the latency listener in the + /// client: it owns the frame id the markers belong to. + /// + private ILatencyStageListener? ActiveLatencyStageListener() => LatencyStageListener ?? device; + + public override void BeginRenderStage(EnumRenderStage stage) + { + // Before the stage's own work: the markers say where rendering began, and every + // stage asks, because which stage comes first is the client's business, not the + // renderer's. + ActiveLatencyStageListener()?.OnFrameRenderStart(); + CurrentRenderStage = stage; + InRenderStage = true; + RenderStageListener?.OnBeginRenderStage(stage); + } + + public override void EndRenderStage(EnumRenderStage stage) + { + // Phase 5: the passes mods declared for this slot run after the stage's renderers, + // still inside the stage (VulkanClientPlatform.ModPasses.cs). + RunModPasses(stage); + InRenderStage = false; + RenderStageListener?.OnEndRenderStage(stage); + } +} + +// Vulkan-native plan, Phase 1A step 4: the device halves of the TAA motion windows and +// the FSR target selection, moved out of ClientPlatformWindows. The guards and the +// window state stay in the base (BeginMotionWrite, EndMotionWrite, BeginMotionOnlyWrite, +// BlitPrimaryToDefault); these are the calls the base makes once a window opens. +public partial class VulkanClientPlatform +{ + /// Primary's default colour set plus the motion attachment. + public override void EnableMotionDrawBuffers() + { + StateDrawBuffers(FrameBuffers[0].FboId, (1 << (MotionAttachmentIndex + 1)) - 1); + } + + /// + /// Back to Primary's default colour set. MotionAttachmentIndex is also the size of + /// the default set (2 without the SSAO G-buffer, 4 with it), because the attachment + /// was appended after it. + /// + public override void RestorePrimaryDrawBuffers() + { + StateDrawBuffers(FrameBuffers[0].FboId, (1 << MotionAttachmentIndex) - 1); + } + + /// The motion attachment alone; the device takes the mask directly. + public override void EnableMotionOnlyDrawBuffers() + { + StateDrawBuffers(FrameBuffers[0].FboId, 1 << MotionAttachmentIndex); + } + + /// Replace-blending on the motion attachment (TAA P3). + public override void ApplyOptimumMotionBlendState() + { + if (!OptimumMotionWriteActive || MotionAttachmentIndex < 0) return; + StateSlotBlend(MotionAttachmentIndex, 32774, 1, 0, 1, 0); + } + + /// Additive (ONE, ONE) blending on the motion attachment for the OIT merge (TAA P4). + public override void ApplyOptimumMotionAccumulateBlendState() + { + if (!OptimumMotionWriteActive || MotionAttachmentIndex < 0) return; + StateSlotBlend(MotionAttachmentIndex, 32774, 1, 1, 1, 1); + } + + /// The FSR EASU pass writes colour attachment 0 of the FSR target. + public override void SelectFsrDrawBuffer(FrameBufferRef target) + { + StateDrawBuffers(target.FboId, 1); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.FrameBuffers.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.FrameBuffers.cs new file mode 100644 index 00000000..6179ba61 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.FrameBuffers.cs @@ -0,0 +1,729 @@ +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using OpenTK.Windowing.Desktop; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: framebuffers and the post chain. The framebuffer +// set, the mod-facing framebuffer factory and disposal are whole overrides (moved from +// ClientPlatformWindows.SetupOptimumFrameBuffers, CreateOptimumFramebuffer and the device +// branches). The binding, clear and pass-state fragments are overrides of the virtuals the +// base's shared post-chain logic calls. +public partial class VulkanClientPlatform +{ + // ClientPlatformWindows' private slot constants; the parity dump and the post chain + // index FrameBuffers by the same numbers. + private const int OptimumFsrFramebufferIndex = 18; + private const int OptimumTaaHistoryIndexA = 19; + private const int OptimumTaaHistoryIndexB = 20; + private const int OptimumTaaSharpenIndex = 21; + private const int OptimumGlR32f = 0x822E; + + // GL keeps the clear colour in driver state and applies it at glClear; the device + // takes it as an argument, so GlClearColorRgbaf records it here and the Default clear + // passes it on. All-zero default, which is what GL_COLOR_CLEAR_VALUE starts as. + private float clearR; + private float clearG; + private float clearB; + private float clearA; + + /// + /// Builds the same framebuffer set as the GL path, through the device. + /// + /// A separate body rather than routed calls inside the GL one, because that body is + /// four hundred lines of raw GL with no seam to route through - it generates its own + /// names and attaches its own textures. Mirroring the layout keeps every index, size + /// and format identical, which is what the render systems assume when they index + /// FrameBuffers by EnumFrameBuffer. + /// + public override List SetupDefaultFrameBuffers() + { + OptimumAdoptFrameBufferSettings(); + bool setupSsao = ClientSettings.SSAOQuality > 0; + List list = new List(31); + for (int i = 0; i <= 24; i++) + { + list.Add(null); + } + int shadowMapQuality = ClientSettings.ShadowMapQuality; + float ssaaLevel = ClientSettings.SSAA; + + int width = (int)((float)((NativeWindow)window).ClientSize.X * ssaaLevel); + int height = (int)((float)((NativeWindow)window).ClientSize.Y * ssaaLevel); + if (width == 0 || height == 0) + { + return list; + } + + bool taaRequested = OptimumTaaRequested; + int motionAttachmentIndex = -1; + + // Primary: depth, colour, glow, and the SSAO position/normal G-buffer. + FrameBufferRef primary = new FrameBufferRef(); + primary.Width = width; + primary.Height = height; + primary.FboId = device.CreateFramebuffer(width, height); + primary.DepthTextureId = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + SetupOptimumTextureSampler(primary.DepthTextureId, 9728, 33071); + device.AttachTexture(primary.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + + int primaryAttachments = (setupSsao ? 4 : 2); + primary.ColorTextureIds = new int[primaryAttachments]; + primary.ColorTextureIds[0] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + primary.ColorTextureIds[1] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + if (setupSsao) + { + primary.ColorTextureIds[2] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + primary.ColorTextureIds[3] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + } + // Match the GL Primary filters, including linear G-buffer sampling and + // the white border used when SSAO projects a sample off screen. + for (int attachment = 0; attachment < primaryAttachments; attachment++) + { + int textureId = primary.ColorTextureIds[attachment]; + SetupOptimumTextureSampler(textureId, + attachment >= 2 || ssaaLevel > 1f ? 9729 : 9728, attachment >= 2 ? 33069 : 10497); + if (attachment >= 2) device.SetTextureBorderColor(textureId, 1f, 1f, 1f, 1f); + } + if (taaRequested) + { + // Optimum: TAA motion attachment, appended after the SSAO G-buffer + // so every existing attachment index is unchanged. Deliberately not + // folded into the draw-buffer mask below - it stays out of every + // pass's output set until a writer opts in (P3+). + try + { + int motionTextureId = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + int[] extendedColorIds = new int[primary.ColorTextureIds.Length + 1]; + Array.Copy(primary.ColorTextureIds, extendedColorIds, primary.ColorTextureIds.Length); + motionAttachmentIndex = primary.ColorTextureIds.Length; + extendedColorIds[motionAttachmentIndex] = motionTextureId; + primary.ColorTextureIds = extendedColorIds; + } + catch (Exception error) + { + DisableOptimumTaa("Primary motion attachment (device): " + error.Message); + motionAttachmentIndex = -1; + } + } + for (int attachment = 0; attachment < primary.ColorTextureIds.Length; attachment++) + { + device.AttachTexture(primary.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + attachment), + primary.ColorTextureIds[attachment], 0); + } + StateDrawBuffers(primary.FboId, (1 << primaryAttachments) - 1); + list[0] = primary; + SetOptimumMotionAttachmentIndex(motionAttachmentIndex); + + // Transparent: OIT accumulation, revealage, glow. Shares Primary's depth. + FrameBufferRef transparent = new FrameBufferRef(); + transparent.Width = width; + transparent.Height = height; + transparent.FboId = device.CreateFramebuffer(width, height); + transparent.ColorTextureIds = new int[3]; + transparent.ColorTextureIds[0] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + transparent.ColorTextureIds[1] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.R16f, EnumTexturePixelFormat.Red, IntPtr.Zero, false); + transparent.ColorTextureIds[2] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + for (int attachment = 0; attachment < 3; attachment++) + { + SetupOptimumTextureSampler(transparent.ColorTextureIds[attachment], 9729, 10497); + device.AttachTexture(transparent.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + attachment), + transparent.ColorTextureIds[attachment], 0); + } + device.AttachTexture(transparent.FboId, EnumFramebufferAttachment.DepthAttachment, primary.DepthTextureId, 0); + StateDrawBuffers(transparent.FboId, 7); + transparent.DepthTextureId = primary.DepthTextureId; + list[1] = transparent; + + if (setupSsao) + { + int ssaoWidth = (int)((float)width * 0.5f); + int ssaoHeight = (int)((float)height * 0.5f); + + FrameBufferRef ssao = new FrameBufferRef(); + ssao.Width = ssaoWidth; + ssao.Height = ssaoHeight; + ssao.FboId = device.CreateFramebuffer(ssaoWidth, ssaoHeight); + ssao.ColorTextureIds = new int[2]; + // GL_RGB in the vanilla path; the device promotes it, because RGB is + // not a guaranteed colour-attachment format in Vulkan. + // A post-chain transient (Transient pool class); see TransientAllocator.PostChainSlots. + ssao.ColorTextureIds[0] = device.CreateTransientTexture2DRaw(ssaoWidth, ssaoHeight, 6407, 13); + device.AttachTexture(ssao.FboId, EnumFramebufferAttachment.ColorAttachment0, ssao.ColorTextureIds[0], 0); + StateDrawBuffers(ssao.FboId, 1); + + // Rotation noise, and the sample kernel that goes with it. Same seed + // and draw order as the GL path, so the pattern matches exactly. + Random random = new Random(5); + int noiseSize = 16; + float[] noise = BuildOptimumSsaoNoise(random, noiseSize); + GCHandle noiseHandle = GCHandle.Alloc(noise, GCHandleType.Pinned); + // GL_RGBA32F, the same internal format the GL path allocates; GL + // uploads GL_RGB data into it and fills alpha with 1 (see + // BuildOptimumSsaoNoise), the device copies all four channels as given. + ssao.ColorTextureIds[1] = device.CreateTexture2DRaw( + noiseSize, noiseSize, 34836, noiseHandle.AddrOfPinnedObject(), 16); + noiseHandle.Free(); + device.SetTextureParameter(ssao.ColorTextureIds[1], + Vintagestory.API.Config.OptimumGlConstants.TextureWrapS, + Vintagestory.API.Config.OptimumGlConstants.Repeat); + device.SetTextureParameter(ssao.ColorTextureIds[1], + Vintagestory.API.Config.OptimumGlConstants.TextureWrapT, + Vintagestory.API.Config.OptimumGlConstants.Repeat); + + float[] ssaoKernel = OptimumSsaoKernel; + for (int sample = 0; sample < 64; sample++) + { + Vec3f kernel = new Vec3f((float)random.NextDouble() * 2f - 1f, (float)random.NextDouble() * 2f - 1f, (float)random.NextDouble()); + kernel.Normalize(); + kernel *= (float)random.NextDouble(); + float scale = (float)sample / 64f; + scale = GameMath.Lerp(0.1f, 1f, scale * scale); + kernel *= scale; + ssaoKernel[sample * 3] = kernel.X; + ssaoKernel[sample * 3 + 1] = kernel.Y; + ssaoKernel[sample * 3 + 2] = kernel.Z; + } + list[13] = ssao; + + list[14] = CreateOptimumColorTarget(ssaoWidth, ssaoHeight, EnumTextureInternalFormat.Rgba8); + list[15] = CreateOptimumColorTarget(ssaoWidth, ssaoHeight, EnumTextureInternalFormat.Rgba8); + } + + list[2] = CreateOptimumColorTarget(width / 2, height / 2, EnumTextureInternalFormat.Rgba8); + list[3] = CreateOptimumColorTarget(width / 2, height / 2, EnumTextureInternalFormat.Rgba8); + list[9] = CreateOptimumColorTarget(width / 4, height / 4, EnumTextureInternalFormat.Rgba8); + list[8] = CreateOptimumColorTarget(width / 4, height / 4, EnumTextureInternalFormat.Rgba8); + list[4] = CreateOptimumColorTarget(width, height, EnumTextureInternalFormat.Rgba16f); + list[7] = CreateOptimumColorTarget(width / 2, height / 2, EnumTextureInternalFormat.Rgba16f); + list[10] = CreateOptimumColorTarget(width, height, EnumTextureInternalFormat.Rgba16f); + + // Optimum: TAA history, render-resolution like Primary. Two slots so the + // resolve reads last frame's parity while writing this frame's; never + // cleared per frame (ClearFrameBuffer(Primary) only touches Primary). + if (taaRequested) + { + try + { + list[OptimumTaaHistoryIndexA] = CreateOptimumHistoryTarget(width, height); + list[OptimumTaaHistoryIndexB] = CreateOptimumHistoryTarget(width, height); + } + catch (Exception error) + { + DisableOptimumTaa("history targets (device): " + error.Message); + list[OptimumTaaHistoryIndexA] = null; + list[OptimumTaaHistoryIndexB] = null; + } + // Optimum TAA (P5): the sharpen target. Its own try - a sharpen + // target that cannot be allocated costs the sharpening, not TAA, + // so it nulls the slot instead of calling DisableOptimumTaa. + try + { + list[OptimumTaaSharpenIndex] = CreateOptimumColorTarget(width, height, + EnumTextureInternalFormat.Rgba16f); + } + catch (Exception error) + { + Logger.Error("Optimum disabled the TAA sharpen pass: {0}", error.Message); + list[OptimumTaaSharpenIndex] = null; + } + } + OptimumAdoptTaaTargets(list, taaRequested); + + // FSR renders at a reduced scale and resolves into a native-sized target. + if (ClientSettings.OptimumRenderScale < 1.0f) + { + list[OptimumFsrFramebufferIndex] = CreateOptimumColorTarget( + ((NativeWindow)window).ClientSize.X, ((NativeWindow)window).ClientSize.Y, + EnumTextureInternalFormat.Rgba8); + } + + list[5] = CreateOptimumDepthTarget(width / 4, height / 4); + + // Both shadow slots always hold a FrameBufferRef, exactly as the GL path + // does: vanilla constructs the objects unconditionally and only allocates + // their textures when the quality setting reaches each level. + // + // The distinction matters because ShaderProgramBase.Use dereferences both + // FrameBuffers[11] and FrameBuffers[12] whenever shadowmapQuality > 0, + // and every shader including fogandlight.fsh - sky.fsh among them - takes + // that branch. Leaving slot 12 null at quality 1 is a null reference on + // the first sky draw, which is what it was. + int shadowSize = Math.Max(4, shadowMapQuality + 2) * 1024; + list[11] = shadowMapQuality > 0 + ? CreateOptimumDepthTarget(shadowSize, shadowSize) + : CreateOptimumPlaceholderTarget(shadowSize, shadowSize); + list[12] = shadowMapQuality > 1 + ? CreateOptimumDepthTarget(shadowSize, shadowSize) + : CreateOptimumPlaceholderTarget(shadowSize, shadowSize); + + for (int shadow = 11; shadow <= 12; shadow++) + { + int textureId = list[shadow].DepthTextureId; + if (textureId == 0) continue; + SetupOptimumTextureSampler(textureId, 9729, 33069); + device.SetTextureBorderColor(textureId, 1f, 1f, 1f, 1f); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureCompareMode, OptimumGlConstants.CompareRefToTexture); + } + + // The post chain's colour textures are transients; record the slot each one serves. + foreach (int transientSlot in Graph.TransientAllocator.PostChainSlots) + { + FrameBufferRef transientTarget = list[transientSlot]; + if (transientTarget == null || transientTarget.ColorTextureIds == null || + transientTarget.ColorTextureIds.Length == 0) continue; + device.OptInTransient(transientTarget.ColorTextureIds[0], transientSlot); + } + + // World/UI separation: the HUD-less scene snapshot and the UI image (UiSeparation.cs). + AllocateUiSeparationTargets(list, width, height); + + OptimumFinishDeviceFrameBufferSetup(list); + return list; + } + + /// + /// The mod-facing framebuffer factory. Same shape as the GL body: create the target, + /// create or adopt a texture per attachment, attach it, then select the colour + /// attachments as draw buffers. + /// + /// The GL body interleaves texture creation with framebuffer attachment through the + /// bound texture unit, and there is no seam to route call-by-call. The attachment order + /// is preserved because the draw-buffer mask is positional: bit N means + /// ColorAttachmentN, and the shaders' output locations depend on it. + /// + public override FrameBufferRef CreateFramebuffer(FramebufferAttrs fbAttrs) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = fbAttrs.Width; + target.Height = fbAttrs.Height; + target.FboId = device.CreateFramebuffer(fbAttrs.Width, fbAttrs.Height); + + List colorTextureIds = new List(); + int drawBufferMask = 0; + FramebufferAttrsAttachment[] attachments = fbAttrs.Attachments; + for (int i = 0; i < attachments.Length; i++) + { + FramebufferAttrsAttachment attachment = attachments[i]; + RawTexture texture = attachment.Texture; + int textureId = texture.TextureId; + if (textureId == 0) + { + textureId = device.CreateTexture2D(texture.Width, texture.Height, + texture.PixelInternalFormat, texture.PixelFormat, IntPtr.Zero, false); + // EnumTextureFilter and EnumTextureWrap carry the GL token values, + // which is exactly what SetTextureParameter expects. + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMinFilter, (int)texture.MinFilter); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMagFilter, (int)texture.MagFilter); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapS, (int)texture.WrapS); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapT, (int)texture.WrapT); + texture.TextureId = textureId; + } + device.AttachTexture(target.FboId, attachment.AttachmentType, textureId, 0); + if (attachment.AttachmentType == EnumFramebufferAttachment.DepthAttachment) + { + target.DepthTextureId = textureId; + } + else + { + colorTextureIds.Add(textureId); + drawBufferMask |= 1 << ((int)attachment.AttachmentType - (int)EnumFramebufferAttachment.ColorAttachment0); + } + } + + target.ColorTextureIds = colorTextureIds.ToArray(); + StateDrawBuffers(target.FboId, drawBufferMask); + + string status; + if (!device.CheckFramebufferComplete(target.FboId, out status)) + { + throw new Exception("FBO " + fbAttrs.Name + ": " + status); + } + return target; + } + + /// Mirror the GL framebuffer texture's filtering and edge policy. + private void SetupOptimumTextureSampler(int textureId, int filter, int wrap) + { + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMinFilter, filter); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMagFilter, filter); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapS, wrap); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapT, wrap); + } + + /// + /// A single-colour-attachment target, as the post chain uses. Every caller is a post-chain + /// slot, so the colour texture is a transient (Transient pool class, registered with the + /// device's transient allocator; SetupDefaultFrameBuffers tags its slot number). + /// + private FrameBufferRef CreateOptimumColorTarget(int width, int height, EnumTextureInternalFormat format) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = width; + target.Height = height; + target.FboId = device.CreateFramebuffer(width, height); + target.ColorTextureIds = new int[1]; + target.ColorTextureIds[0] = device.CreateTransientTexture2D(width, height, format, -1); + // setupAttachment uses linear filtering and edge clamping. FXAA and + // the reduced-resolution blur passes require fractional texel samples. + SetupOptimumTextureSampler(target.ColorTextureIds[0], 9729, 33071); + device.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + StateDrawBuffers(target.FboId, 1); + return target; + } + + /// A depth-only target, as the shadow maps and liquid depth use. + private FrameBufferRef CreateOptimumDepthTarget(int width, int height) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = width; + target.Height = height; + target.FboId = device.CreateFramebuffer(width, height); + target.ColorTextureIds = new int[0]; + target.DepthTextureId = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + SetupOptimumTextureSampler(target.DepthTextureId, 9729, 33071); + device.AttachTexture(target.FboId, EnumFramebufferAttachment.DepthAttachment, target.DepthTextureId, 0); + StateDrawBuffers(target.FboId, 0); + return target; + } + + /// + /// A TAA history slot - colour (RGBA16F), aux (RGBA8: glow.rg, ssao.b) and linear + /// depth (R32F), MRT-written by the resolve pass and read back next frame. R32F has no + /// entry, so it goes through + /// CreateTexture2DRaw with the raw GL token, the same way the SSAO noise + /// texture does above. + /// + private FrameBufferRef CreateOptimumHistoryTarget(int width, int height) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = width; + target.Height = height; + target.FboId = device.CreateFramebuffer(width, height); + target.ColorTextureIds = new int[3]; + target.ColorTextureIds[0] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + target.ColorTextureIds[1] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + target.ColorTextureIds[2] = device.CreateTexture2DRaw(width, height, OptimumGlR32f, IntPtr.Zero, 4); + // Optimum TAA: the resolve reprojects the history by a fractional pixel + // offset, so colour (Catmull-Rom taps) and glow (a plain bilinear fetch) + // must filter LINEAR; sampling them NEAREST snaps the reprojection to + // whole pixels and the history never converges. Linear depth stays + // NEAREST - interpolating across a silhouette invents a depth that is on + // neither surface and defeats the disocclusion test. Clamp to edge on + // all three, matching CreateOptimumHistoryTargetGl. + SetupOptimumTextureSampler(target.ColorTextureIds[0], 9729, 33071); + SetupOptimumTextureSampler(target.ColorTextureIds[1], 9729, 33071); + SetupOptimumTextureSampler(target.ColorTextureIds[2], 9728, 33071); + for (int attachment = 0; attachment < 3; attachment++) + { + device.AttachTexture(target.FboId, + (EnumFramebufferAttachment)((int)EnumFramebufferAttachment.ColorAttachment0 + attachment), + target.ColorTextureIds[attachment], 0); + } + StateDrawBuffers(target.FboId, 7); + if (!device.CheckFramebufferComplete(target.FboId, out string status)) + { + throw new Exception("Optimum TAA history FBO: " + status); + } + return target; + } + + /// + /// A framebuffer slot that exists but owns nothing, for a quality level whose + /// resources are not allocated. + /// + /// The GL path leaves such a slot holding a FrameBufferRef whose ids are zero; callers + /// read its Width and Height and bind its texture id, and binding zero is a no-op + /// there. The device treats texture 0 as unbound and substitutes its placeholder, so + /// the same read is equally harmless here. + /// + private static FrameBufferRef CreateOptimumPlaceholderTarget(int width, int height) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = width; + target.Height = height; + target.ColorTextureIds = new int[0]; + return target; + } + + public override void DisposeFrameBuffer(FrameBufferRef frameBuffer, bool disposeTextures = true) + { + if (frameBuffer == null) + { + return; + } + if (disposeTextures) + { + for (int i = 0; i < frameBuffer.ColorTextureIds.Length; i++) + { + GLDeleteTexture(frameBuffer.ColorTextureIds[i]); + } + if (frameBuffer.DepthTextureId > 0) + { + GLDeleteTexture(frameBuffer.DepthTextureId); + } + } + // GLDeleteTexture above already routes, so only the target itself is left. + stated.ForgetFramebuffer(frameBuffer.FboId); + device.DeleteFramebuffer(frameBuffer.FboId); + } + + /// + /// SetupDefaultFrameBuffers shares one depth texture between Primary and Transparent, + /// so the same handle appears in more than one FrameBufferRef. Deleting it twice + /// double-frees and makes VulkanStats.NoteTextureDeleted over-count, so every handle + /// is deleted once. + /// + public override void DisposeFrameBuffers(List buffers) + { + // The UI image may be what Default resolves to; it is about to go. + CloseUiScope(); + // The AO targets are sized to Primary and go with it. + ReleaseAmbientOcclusionTargets(); + HashSet deletedTextures = new HashSet(); + for (int k = 0; k < buffers.Count; k++) + { + if (buffers[k] != null) + { + stated.ForgetFramebuffer(buffers[k].FboId); + device.DeleteFramebuffer(buffers[k].FboId); + if (deletedTextures.Add(buffers[k].DepthTextureId)) + { + device.DeleteTexture(buffers[k].DepthTextureId); + } + for (int n = 0; n < buffers[k].ColorTextureIds.Length; n++) + { + if (deletedTextures.Add(buffers[k].ColorTextureIds[n])) + { + device.DeleteTexture(buffers[k].ColorTextureIds[n]); + } + } + buffers[k].Disposed = true; + } + } + } + + /// + /// Swaps the colour attachment on an already-created target, which mods use to render + /// into a texture they own. + /// + public override void LoadFrameBuffer(FrameBufferRef frameBuffer, int textureId) + { + CurrentFrameBuffer = frameBuffer; + device.AttachTexture(frameBuffer.FboId, EnumFramebufferAttachment.ColorAttachment0, textureId, 0); + } + + /// + /// GL keeps a clear colour in its state; the device takes it at clear time, so this + /// only records what the next Default clear should use. + /// + public override void GlClearColorRgbaf(float r, float g, float b, float a) + { + clearR = r; + clearG = g; + clearB = b; + clearA = a; + } + + /// + /// FboId carries the device's render-target handle on this path, the same way + /// VAO.VaoId carries a mesh handle, so FrameBufferRef stays the type mods already hold. + /// + public override void BindCurrentFrameBuffer(FrameBufferRef value) + { + // No device bind: every draw and clear names its target (CurrentFrameBuffer). The GL body + // also sets the viewport to the whole target, which is stated here. The latest bind wins, + // so a raw fork bind before it no longer addresses the draws. + forkFramebuffer = 0; + if (value != null) NoteForkViewport(0, 0, value.Width, value.Height); + } + + public override void BindCurrentFrameBufferKeepViewport(FrameBufferRef value) + { + forkFramebuffer = 0; + } + + public override void ClearBoundFrameBuffer(FrameBufferRef framebuffer, float[] clearColor, bool clearDepthBuffer, bool clearColorBuffers) + { + if (clearColorBuffers) + { + for (int k = 0; k < framebuffer.ColorTextureIds.Length; k++) + { + ClearTargetColor(framebuffer.FboId, k, clearColor[0], clearColor[1], clearColor[2], clearColor[3]); + } + } + if (clearDepthBuffer) + { + ClearTargetDepth(framebuffer.FboId, 1f); + } + } + + /// + /// Same clear values per pass as the GL body. Default clears the swapchain image with + /// the colour GlClearColorRgbaf recorded, since the device has no GL clear-colour state + /// of its own. + /// + public override void ClearFrameBufferPass(EnumFrameBuffer framebuffer) + { + int target = CurrentTargetId; + switch (framebuffer) + { + case EnumFrameBuffer.Default: + ClearTargetColor(target, 0, clearR, clearG, clearB, clearA); + ClearTargetDepth(target, 1f); + break; + case EnumFrameBuffer.Primary: + ClearTargetColor(target, 0, 0f, 0f, 0f, 1f); + ClearTargetColor(target, 1, 0f, 0f, 0f, 1f); + if (OptimumRenderSsao) + { + ClearTargetColor(target, 2, 0f, 0f, 0f, 1f); + ClearTargetColor(target, 3, 0f, 0f, 0f, 1f); + } + if (MotionAttachmentIndex >= 0) + { + // A clear honours the draw buffers, as on GL. + // Motion is excluded until a writer opts in, so temporarily + // enable it just as the GL branch does. Otherwise stale + // motion/reactivity survives and can reject all TAA history. + StateDrawBuffers(FrameBuffers[0].FboId, (1 << (MotionAttachmentIndex + 1)) - 1); + ClearTargetColor(target, MotionAttachmentIndex, 0f, 0f, 0f, 0f); + StateDrawBuffers(FrameBuffers[0].FboId, (1 << MotionAttachmentIndex) - 1); + } + ClearTargetDepth(target, 1f); + break; + case EnumFrameBuffer.LiquidDepth: + case EnumFrameBuffer.ShadowmapFar: + case EnumFrameBuffer.ShadowmapNear: + { + FrameBufferRef optimumTarget = FrameBuffers[(int)framebuffer]; + NoteForkViewport(0, 0, optimumTarget.Width, optimumTarget.Height); + ClearTargetDepth(target, 1f); + break; + } + case EnumFrameBuffer.Transparent: + // Weighted-blended OIT: accumulation starts at zero, revealage at + // one, and the third attachment is the opaque-depth copy. + ClearTargetColor(target, 0, 0f, 0f, 0f, 0f); + ClearTargetColor(target, 1, 1f, 0f, 0f, 0f); + ClearTargetColor(target, 2, 0f, 0f, 0f, 0f); + break; + } + } + + /// + /// Weighted-blended OIT: accumulation adds, revealage multiplies, and the third + /// attachment uses ordinary source-alpha blending. + /// + public override void ApplyTransparentPassBlendState() + { + StateDrawBuffers(FrameBuffers[1].FboId, 7); + StateBlend(true, EnumBlendMode.Standard); + StateSlotBlend(0, 32774, 1, 1, 1, 1); + StateSlotBlend(1, 32774, 0, 769, 0, 769); + StateSlotBlend(2, 32774, 770, 771, 770, 771); + // Phase 3b stage 2: the same contract, recorded for the native chunk passes that draw + // into this target (VulkanClientPlatform.NativeChunks.cs). + NoteNativeTransparentBlend(0, 32774, 1, 1, 1, 1); + NoteNativeTransparentBlend(1, 32774, 0, 769, 0, 769); + NoteNativeTransparentBlend(2, 32774, 770, 771, 770, 771); + nativeTransparentSlots = 7; + } + + /// + /// Selecting GL_BACK has no device equivalent: binding the default target already + /// means the swapchain image. + /// + public override void SelectBackDrawBuffer() + { + } + + public override void SetBlendEnabled(bool enabled) + { + stated.SetBlendEnabled(enabled); + statedBlendOn = enabled; + } + + /// + /// The OIT merge's three state calls, spelled out rather than routed through + /// GlToggleBlend, because that helper also overrides the SSAO attachments and this + /// pass deliberately sets only the global mode. + /// + public override void ApplyTransparentMergeBlendState() + { + GlDisableDepthTest(); + StateBlend(true, EnumBlendMode.Standard); + StateSlotBlendFunc(0, 770, 771, 770, 771); + } + + /// + /// The SSAO rotation noise as RGBA float texels, drawing from + /// in the GL path's order (two doubles per texel) so the sample kernel drawn after it + /// matches too. + /// + /// Alpha is 1, not 0. The GL path allocates GL_RGBA32F and uploads GL_RGB pixel data; + /// GL's pixel transfer fills the missing alpha with 1, so the texture holds 1.0 in every + /// texel (the parity dump reads 1.0 on OpenGL). The device takes the four channels + /// verbatim, and a padding 0 here left SSAO colour1 alpha at 0.0 on Vulkan. + /// + internal static float[] BuildOptimumSsaoNoise(Random random, int noiseSize) + { + float[] noise = new float[noiseSize * noiseSize * 4]; + Vec3f direction = new Vec3f(); + for (int texel = 0; texel < noiseSize * noiseSize; texel++) + { + direction.Set((float)random.NextDouble() * 2f - 1f, (float)random.NextDouble() * 2f - 1f, 0f).Normalize(); + noise[texel * 4] = direction.X; + noise[texel * 4 + 1] = direction.Y; + noise[texel * 4 + 2] = direction.Z; + noise[texel * 4 + 3] = 1f; + } + return noise; + } + + public override void ClearSsaoTarget() + { + ClearTargetColor(CurrentTargetId, 0, 1f, 1f, 1f, 1f); + } + + /// + /// The device takes the target explicitly (the bound one, Primary here), and its mask + /// is positional - bit N selects ColorAttachmentN. + /// + public override void BeginFinalCompositionDrawBuffers() + { + StateDrawBuffers(CurrentFrameBuffer != null ? CurrentFrameBuffer.FboId : 0, 1); + GlDisableDepthTest(); + } + + public override void RestoreWorldDrawBuffers(bool ssaoAttachments) + { + if (ssaoAttachments) + { + StateDrawBuffers(CurrentFrameBuffer != null ? CurrentFrameBuffer.FboId : 0, 15); + } + else + { + StateDrawBuffers(CurrentFrameBuffer != null ? CurrentFrameBuffer.FboId : 0, 3); + } + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Graph.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Graph.cs new file mode 100644 index 00000000..fb994b8b --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Graph.cs @@ -0,0 +1,382 @@ +using System.Collections.Generic; +using System.Globalization; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 2 step 2, as it stands after the emulation layer went: every pass is +// declared by the native route that records it (BeginNativePass), in frame order. The pass +// context below is only the naming prefix and flag set of the stage or post method running - +// the render stage (from the C3 bracket), or the OIT merge, TAA resolve and sharpen, +// post-processing, final composition, blit, sky motion and liquid motion - which the entity +// route names its coalesced pass after. Binds declare nothing. +public partial class VulkanClientPlatform +{ + // ClientPlatformWindows' EnumFrameBuffer slots the post chain indexes. + private const int PrimaryIndex = 0; + private const int TransparentIndex = 1; + private const int BlurHorizontalMedResIndex = 2; + private const int BlurVerticalMedResIndex = 3; + private const int FindBrightIndex = 4; + private const int LiquidDepthIndex = 5; + private const int GodRaysIndex = 7; + private const int BlurVerticalLowResIndex = 8; + private const int BlurHorizontalLowResIndex = 9; + private const int LumaIndex = 10; + private const int ShadowFarIndex = 11; + private const int ShadowNearIndex = 12; + private const int SsaoIndex = 13; + private const int SsaoBlurVerticalIndex = 14; + private const int SsaoBlurHorizontalIndex = 15; + + private string passContext = "Frame"; + private PassFlags passContextFlags = PassFlags.AllowSplit; + + /// Forwards the render-stage bracket to the pass declarations. + private sealed class FrameGraphStageListener : IRenderStageListener + { + private readonly VulkanClientPlatform platform; + + public FrameGraphStageListener(VulkanClientPlatform platform) => this.platform = platform; + + public void OnBeginRenderStage(EnumRenderStage stage) + { + platform.SetPassContext(stage.ToString(), StageFlags(stage)); + } + + public void OnEndRenderStage(EnumRenderStage stage) + { + platform.GraphDevice?.EndStagePass(); + // The liquid motion pass runs right after the AfterOIT renderers (ClientMain.MainRenderLoop). + platform.SetPassContext(stage == EnumRenderStage.AfterOIT ? "LiquidMotion" : "Frame", PassFlags.AllowSplit); + } + } + + private VulkanDevice? GraphDevice => device; + + /// World stages draw declared targets; everything a mod hosts may sample anything. + internal static PassFlags StageFlags(EnumRenderStage stage) => stage switch + { + EnumRenderStage.Before or EnumRenderStage.ShadowFar or EnumRenderStage.ShadowFarDone or + EnumRenderStage.ShadowNear or EnumRenderStage.ShadowNearDone or EnumRenderStage.Opaque or + EnumRenderStage.OIT => PassFlags.AllowSplit, + _ => PassFlags.OpenSampling | PassFlags.AllowSplit, + }; + + /// Starts a context: the prefix and flags of the passes recorded under it. + private void SetPassContext(string context, PassFlags flags) + { + passContext = context; + passContextFlags = flags; + } + + /// + /// The (context, current target) pass name. The entity route records every entity of a stage + /// under it with the scope kept open, so RenderTargetManager.DeclarePass coalesces instead of + /// ending the rendering scope and starting another per entity. + /// + private string BoundPassName() + { + int id = CurrentTargetId; + int index = FrameBufferIndexOf(id); + string target = index >= 0 + ? index.ToString(CultureInfo.InvariantCulture) + : id == PassDeclaration.DefaultFramebuffer + ? "Default" + : "fbo" + id.ToString(CultureInfo.InvariantCulture); + return passContext + "/" + target; + } + + private int FrameBufferIndexOf(int framebufferId) + { + List buffers = FrameBuffers; + if (buffers == null || framebufferId <= 0) return -1; + for (int i = 0; i < buffers.Count; i++) + { + if (buffers[i] != null && buffers[i].FboId == framebufferId) return i; + } + return -1; + } + + /// + /// The textures the base's pass body samples for (context, target). Where the base picks + /// one of several (the resolved scene, either history parity) all candidates + /// are listed: the set stays the same from frame to frame, which the plan needs. + /// + internal int[] PassReads(string context, int target) + { + var reads = new List(); + switch (context) + { + case "MergeTransparent": + AddColour(reads, TransparentIndex, 0); + AddColour(reads, TransparentIndex, 1); + AddColour(reads, TransparentIndex, 2); + break; + case "SkyMotion": + AddColour(reads, TransparentIndex, 1); + break; + case "TaaResolve": + AddColour(reads, PrimaryIndex, 0); + AddColour(reads, PrimaryIndex, 1); + // Absent motion attachment: the index stays -1. + if (MotionAttachmentIndex > -1) AddColour(reads, PrimaryIndex, MotionAttachmentIndex); + AddDepth(reads, PrimaryIndex); + for (int slot = 0; slot < 3; slot++) + { + AddColour(reads, OptimumTaaHistoryIndexA, slot); + AddColour(reads, OptimumTaaHistoryIndexB, slot); + } + break; + case "TaaSharpen": + // Sharpen runs after FinalComposition and samples the composited Primary colour, + // not either history slot that fed the resolve. + AddColour(reads, PrimaryIndex, 0); + break; + case "Post": + switch (target) + { + case FindBrightIndex: + case GodRaysIndex: + case LumaIndex: + AddPostScene(reads); + break; + case BlurHorizontalMedResIndex: + AddColour(reads, FindBrightIndex, 0); + break; + case BlurVerticalMedResIndex: + AddColour(reads, BlurHorizontalMedResIndex, 0); + break; + case BlurHorizontalLowResIndex: + AddColour(reads, BlurVerticalMedResIndex, 0); + break; + case BlurVerticalLowResIndex: + AddColour(reads, BlurHorizontalLowResIndex, 0); + break; + case SsaoIndex: + AddColour(reads, PrimaryIndex, 2); + AddColour(reads, PrimaryIndex, 3); + AddColour(reads, SsaoIndex, 1); + AddColour(reads, TransparentIndex, 1); + break; + case SsaoBlurHorizontalIndex: + AddColour(reads, SsaoIndex, 0); + AddColour(reads, SsaoBlurVerticalIndex, 0); + AddDepth(reads, PrimaryIndex); + break; + case SsaoBlurVerticalIndex: + AddColour(reads, SsaoBlurHorizontalIndex, 0); + break; + } + break; + case "FinalComposition": + AddColour(reads, LumaIndex, 0); + AddColour(reads, BlurVerticalLowResIndex, 0); + AddColour(reads, GodRaysIndex, 0); + AddColour(reads, PrimaryIndex, 1); + AddColour(reads, OptimumTaaHistoryIndexA, 1); + AddColour(reads, OptimumTaaHistoryIndexB, 1); + AddColour(reads, SsaoBlurVerticalIndex, 0); + break; + case "Blit": + AddColour(reads, PrimaryIndex, 0); + AddColour(reads, OptimumFsrFramebufferIndex, 0); + AddColour(reads, OptimumTaaSharpenIndex, 0); + break; + case "Before": + case "ShadowFar": + case "ShadowNear": + case "Opaque": + case "OIT": + AddDepth(reads, ShadowFarIndex); + AddDepth(reads, ShadowNearIndex); + AddDepth(reads, LiquidDepthIndex); + break; + } + return reads.ToArray(); + } + + /// + /// Slots a pass overwrites completely with blending off and never reads from an earlier + /// frame: the bloom blur chain and the SSAO blur targets. The plan may load them DONT_CARE. + /// + internal static uint PassTransientSlots(string context, int target) + { + if (context != "Post") return 0; + return target switch + { + FindBrightIndex or BlurHorizontalMedResIndex or BlurVerticalMedResIndex or BlurHorizontalLowResIndex or + BlurVerticalLowResIndex or SsaoBlurHorizontalIndex or SsaoBlurVerticalIndex => 1u, + _ => 0u, + }; + } + + /// + /// Every texture the post chain may read as the scene: Primary and the TAA histories. The + /// sharpen output is produced after final composition and its late overlays, then consumed only + /// by the final blit. + /// + private void AddPostScene(List reads) + { + AddColour(reads, PrimaryIndex, 0); + AddColour(reads, PrimaryIndex, 1); + AddColour(reads, OptimumTaaHistoryIndexA, 0); + AddColour(reads, OptimumTaaHistoryIndexA, 1); + AddColour(reads, OptimumTaaHistoryIndexB, 0); + AddColour(reads, OptimumTaaHistoryIndexB, 1); + } + + private void AddColour(List reads, int index, int slot) + { + List buffers = FrameBuffers; + if (buffers == null || index < 0 || index >= buffers.Count) return; + FrameBufferRef buffer = buffers[index]; + if (buffer?.ColorTextureIds == null || slot < 0 || slot >= buffer.ColorTextureIds.Length) return; + int id = buffer.ColorTextureIds[slot]; + if (id > 0 && !reads.Contains(id)) reads.Add(id); + } + + private void AddDepth(List reads, int index) + { + List buffers = FrameBuffers; + if (buffers == null || index < 0 || index >= buffers.Count || buffers[index] == null) return; + int id = buffers[index].DepthTextureId; + if (id > 0 && !reads.Contains(id)) reads.Add(id); + } + + // ------------------------------------------------------------ post methods + + /// + /// Phase 3b stage 1: the chain's first pass, drawn natively + /// (VulkanClientPlatform.NativePostChain.cs). puts the + /// whole chain back on the OpenGL body for the differential tests. + /// + public override void MergeTransparentRenderPass() + { + NotePostStep(NativePostStep.OitMerge); + if (UseNativePostChain) + { + NativeOitMerge(); + return; + } + LegacyOitMerge(); + } + + /// Phase 3b stage 1: the chain's second pass, drawn natively. + public override bool RenderOptimumSkyMotion() + { + NotePostStep(NativePostStep.SkyMotion); + return UseNativePostChain ? NativeSkyMotion() : LegacySkyMotion(); + } + + /// + /// Phase 3b stage 1: Optimum owns the post chain's order. The native route runs the steps + /// this method holds - AO, TAA resolve, bloom, god rays, the Luma step and the epilogue - + /// and never calls base. Sharpen is deliberately at the blit boundary, after late overlays. + /// + public override void RenderPostprocessingEffects(float[] projectMatrix) + { + if (UseNativePostChain) + { + RunNativePostChain(projectMatrix); + return; + } + SetPassContext("Post", PassFlags.None); + base.RenderPostprocessingEffects(projectMatrix); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + public override bool RenderOptimumTaaResolve() + { + string outer = passContext; + PassFlags outerFlags = passContextFlags; + SetPassContext("TaaResolve", PassFlags.None); + bool resolved = base.RenderOptimumTaaResolve(); + SetPassContext(outer, outerFlags); + return resolved; + } + + public override int RenderOptimumTaaSharpen(int resolvedScene) + { + if (UseNativePostChain) NotePostStep(NativePostStep.TaaSharpen); + string outer = passContext; + PassFlags outerFlags = passContextFlags; + SetPassContext("TaaSharpen", PassFlags.None); + int sharpened = base.RenderOptimumTaaSharpen(resolvedScene); + SetPassContext(outer, outerFlags); + return sharpened; + } + + /// + /// Phase 3b stage 1: the resolve's draw, natively. The lib body above keeps every temporal + /// decision and every field it writes afterwards, so only the draw changes route. + /// + public override void OptimumTaaResolveDraw(FrameBufferRef write, FrameBufferRef read, + float[] invViewProjJittered, float[] prevViewProj, bool reset) + { + if (UseNativePostChain) + { + NativeTaaResolve(write, read, invViewProjJittered, prevViewProj, reset); + return; + } + base.OptimumTaaResolveDraw(write, read, invViewProjJittered, prevViewProj, reset); + } + + /// Phase 3b stage 1: the sharpen's draw, natively. + public override void OptimumTaaSharpenDraw(FrameBufferRef target, int resolvedScene) + { + if (UseNativePostChain) + { + NativeTaaSharpen(target, resolvedScene); + return; + } + base.OptimumTaaSharpenDraw(target, resolvedScene); + } + + /// + /// Phase 3b stage 1: the chain's ninth pass, drawn natively - the attachment-subset pass that + /// writes Primary colour 0 while sampling Primary colour 1 (VulkanClientPlatform.NativePostFinal.cs). + /// + public override void RenderFinalComposition() + { + NotePostStep(NativePostStep.FinalComposition); + if (UseNativePostChain) + { + NativeFinalComposition(); + } + else + { + LegacyFinalComposition(); + } + // World/UI separation: the composited image holds the scene alone for exactly this long - + // RenderAfterFinalComposition draws the world-space overlays onto it next. + CaptureSceneNoHud(); + } + + /// + /// Phase 3b: the blit runs natively (VulkanClientPlatform.NativeBlit.cs) - its own + /// pipelines, one declared pass per written target, no GL-shaped call in between. The + /// GL body stays reachable through for the old-route + /// side of the parity tests. + /// + public override void BlitPrimaryToDefault() + { + if (NativeBlitEnabled && UseNativePostChain) + { + RenderNativeBlit(); + } + else + { + NotePostStep(NativePostStep.Blit); + SetPassContext("Blit", PassFlags.None); + base.BlitPrimaryToDefault(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + // World/UI separation: the boundary. Everything ScreenManager draws after this call - the + // AfterBlit stage, the menu background, the Ortho stage - goes into the UI image, on every + // route out of the blit (debug view, FSR, plain, no offscreen buffer). + OpenUiScope(); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs new file mode 100644 index 00000000..4744f392 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs @@ -0,0 +1,243 @@ +using System; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 5: the leaf operations the render systems outside the +// platform used to route through the (now deleted) static device seam - ScreenManager's GUI depth clear, +// ClientMain's depth range, mesh handle deletion, ChunkRenderer's and ShaderRegistry's LOD +// bias, the framebuffer debug overlay's depth compare, SvgLoader's upload, +// InventoryItemRenderer's atlas slot clear, the OIT targets and pass state, the sun +// occlusion probe, Screenshot's readback and the backend name. Each body is the device +// branch the call site had, moved unchanged; ClientPlatformWindows holds the GL lines. +public partial class VulkanClientPlatform +{ + /// + /// glDepthRange clamps both bounds to [0, 1], and every caller passes a range that + /// clamps to the default (0, 20000) or restates it (0, 1). The device's viewport + /// depth range is that default, so there is nothing to do. + /// + public override void SetDepthRange(float near, float far) + { + } + + /// glClearBuffer clamps a depth clear value to [0, 1]; the device takes the clamped value. + public override void ClearDefaultDepth(float depth) + { + ClearTargetDepth(CurrentTargetId, Math.Clamp(depth, 0f, 1f)); + } + + /// The shared index buffer is a device mesh handle on this path. + public override void DeleteMeshHandle(int bufferId) + { + device.DeleteMesh(bufferId); + } + + /// + /// On this path VaoId is the device's mesh handle and the per-attribute buffer fields + /// are zero. VAO.Dispose is the one place a mesh is released, as on GL: the client and + /// mods call MeshRef.Dispose directly at least as often as DeleteMesh (which only + /// disposes the VAO), so the device mesh is released here, through the device's + /// deferred deletion (destroyed once the timelines pass every frame that drew it). + /// VAO.Dispose runs once per VAO (its Disposed guard) and never from the finalizer. + /// After ShutdownGraphics the device is gone and took every mesh with it. + /// + public override void DeleteVertexArrayHandles(VAO vao) + { + if (device != null && vao.VaoId != 0) + { + device.DeleteMesh(vao.VaoId); + } + } + + /// The device addresses each texture directly; nothing to bind or restore. + public override void SetTextureLodBias(int[] textureIds, float bias) + { + for (int k = 0; k < textureIds.Length; k++) + { + device.SetTextureParameter(textureIds[k], OptimumGlConstants.TextureLodBias, bias); + } + } + + public override void SetSamplerLodBias(int samplerId, float bias) + { + device.SetSamplerParameter(samplerId, OptimumGlConstants.TextureLodBias, bias); + } + + /// The device takes the texture itself rather than whatever is bound. + public override void SetTextureDepthCompare(int textureId, int mode) + { + device.SetTextureParameter(textureId, OptimumGlConstants.TextureCompareMode, mode); + } + + /// + /// The device takes the atlas texture by id. The pixels the caller passes are all + /// zero, so channel order does not matter. + /// + public override void ClearTextureRegion(int textureId, int x, int y, int width, int height, int[] pixels) + { + GCHandle pin = GCHandle.Alloc(pixels, GCHandleType.Pinned); + try + { + device.UploadTexture2D(textureId, 0, x, y, width, height, EnumTexturePixelFormat.Rgba, pin.AddrOfPinnedObject()); + } + finally + { + pin.Free(); + } + } + + public override int LoadTextureFromRgbaPointer(int width, int height, IntPtr pixels) + { + int textureId = device.CreateTexture2DRaw(width, height, OptimumGlConstants.Rgba8, pixels, 4); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMagFilter, 9729); + return textureId; + } + + public override void SetProgramSamplerUnit(int programId, string samplerName, int unit) + { + device.SetSamplerUnit(programId, samplerName, unit); + } + + /// + /// The reveal target and the accumulation array, one layer per OIT weight bucket, + /// attached layer by layer at colour attachments 3, 4 and 5. The device addresses + /// textures directly, so there is no framebuffer or texture to bind first. + /// + public override void CreateOitTargets(FrameBufferRef transparent, int layers, out int revealTexture, out int accumTexture) + { + int width = transparent.Width; + int height = transparent.Height; + revealTexture = device.CreateTexture2DRaw(width, height, OptimumGlConstants.Rgba8, IntPtr.Zero, 0); + SetOitSampling(revealTexture); + + accumTexture = device.CreateTexture2DArray(width, height, layers, + EnumTextureInternalFormat.Rgba16f, EnumTexturePixelFormat.Rgba); + SetOitSampling(accumTexture); + + device.AttachTexture(transparent.FboId, EnumFramebufferAttachment.ColorAttachment0, revealTexture, 0); + device.AttachTexture(transparent.FboId, EnumFramebufferAttachment.ColorAttachment3, accumTexture, 0); + device.AttachTexture(transparent.FboId, EnumFramebufferAttachment.ColorAttachment4, accumTexture, 1); + device.AttachTexture(transparent.FboId, (EnumFramebufferAttachment)36069, accumTexture, 2); + } + + /// + /// Nearest filtering and clamped wrapping: both OIT targets are read back per fragment + /// at exactly the coordinate that produced them. + /// + private void SetOitSampling(int textureId) + { + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMinFilter, 9728); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureMagFilter, 9728); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapS, 33071); + device.SetTextureParameter(textureId, OptimumGlConstants.TextureWrapT, 33071); + } + + /// + /// Attachments 0 and 1 hold revealage and multiply down from one; 3, 4 and 5 + /// accumulate additively from zero. Attachment 2 is written by the pass itself and + /// keeps vanilla blending. + /// + public override void BeginOitAccumulation(FrameBufferRef transparent) + { + StateDrawBuffers(transparent.FboId, 0x3F); + StateSlotBlendFunc(0, 774, 0, 774, 0); + StateSlotBlendFunc(1, 774, 0, 774, 0); + StateSlotBlendFunc(3, 1, 1, 1, 1); + StateSlotBlendFunc(4, 1, 1, 1, 1); + StateSlotBlendFunc(5, 1, 1, 1, 1); + // Phase 3b stage 2: the same contract, recorded for the native chunk passes that draw + // into this target - a native pipeline states its blend rather than reading the + // tracker's back (VulkanClientPlatform.NativeChunks.cs). Slot 2 keeps whatever the + // vanilla transparent set left there, exactly as GL does. + NoteNativeTransparentBlend(0, 32774, 774, 0, 774, 0); + NoteNativeTransparentBlend(1, 32774, 774, 0, 774, 0); + NoteNativeTransparentBlend(3, 32774, 1, 1, 1, 1); + NoteNativeTransparentBlend(4, 32774, 1, 1, 1, 1); + NoteNativeTransparentBlend(5, 32774, 1, 1, 1, 1); + nativeTransparentSlots = 0x3F; + ClearTargetColor(transparent.FboId, 0, 1f, 1f, 1f, 1f); + ClearTargetColor(transparent.FboId, 1, 1f, 1f, 1f, 1f); + ClearTargetColor(transparent.FboId, 3, 0f, 0f, 0f, 0f); + ClearTargetColor(transparent.FboId, 4, 0f, 0f, 0f, 0f); + ClearTargetColor(transparent.FboId, 5, 0f, 0f, 0f, 0f); + } + + /// Units 6 and 7; the device binds by unit whatever the texture's dimensionality. + public override void BindOitTextures(int revealTexture, int accumTexture) + { + stated.BindTexture(6, revealTexture); + stated.BindTexture(7, accumTexture); + } + + public override int GenOcclusionQuery() + { + return device.CreateOcclusionQuery(); + } + + public override void BeginOcclusionQuery(int queryId) + { + device.BeginOcclusionQuery(queryId); + } + + public override void EndOcclusionQuery(int queryId) + { + device.EndOcclusionQuery(queryId); + } + + /// + /// GL's form polls availability and reads the sample count only when it is there; the + /// device's reads the same way (Phase 1B replaces the flush inside GetQueryResult). + /// + public override bool TryGetOcclusionQueryResult(int queryId, out int samples) + { + if (device.IsQueryResultAvailable(queryId)) + { + samples = device.GetQueryResult(queryId); + return true; + } + samples = 0; + return false; + } + + public override void DeleteOcclusionQuery(int queryId) + { + device.DeleteQuery(queryId); + } + + /// + /// The device reads back colour attachment 0 of the target the client has current, which is + /// the same image GL would read from the bound framebuffer and in the same orientation - the + /// one flip happens at present, after this. + /// + /// Channel order is converted here. The OpenGL body of this virtual is + /// glReadPixels(..., GL_BGRA, ...), and its callers depend on that: + /// Screenshot.GrabScreenshot - the screenshot key and the AVI recorder - decodes + /// into an SKBitmap declared Bgra8888, and the headless harness writes + /// its frames from the same call. The device's default colour target is + /// R8G8B8A8_UNORM, so without this every Vulkan screenshot and recording comes + /// out with red and blue exchanged. The conversion sits here rather than in + /// VulkanDevice.ReadDefaultFramebuffer because that method is also the + /// backend's general "read the bound target" operation, which the GPU tests use to + /// inspect attachments in their stored order. + /// + /// A target that is already B G R A needs nothing done, which is why the format + /// is asked rather than assumed; is + /// left at the base's true either way, because this method makes it true. + /// + public override void ReadDefaultFramebuffer(int x, int y, int width, int height, IntPtr destination) + { + device.ReadFramebufferColor(CurrentTargetId, x, y, width, height, destination); + if (destination == IntPtr.Zero || width <= 0 || height <= 0) return; + if (device.DefaultColorFormat is Format.B8G8R8A8Unorm or Format.B8G8R8A8Srgb) return; + PixelOrder.SwapRedAndBlue(destination, (long)width * height); + } + + public override string GraphicsBackendName => device.BackendName; +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Meshes.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Meshes.cs new file mode 100644 index 00000000..4a5757aa --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Meshes.cs @@ -0,0 +1,258 @@ +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using OpenTK.Graphics.OpenGL; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: mesh allocation, upload, update and draw. Each body +// is the device branch that opened the same ClientPlatformWindows method, moved unchanged, +// plus the API-neutral lines around it the base method ran on both paths (draw-call +// accounting, argument checks, the SSBO face packing). +// +// On this path VAO.VaoId is the device's mesh handle rather than a GL vertex-array name. +// Reusing the field keeps MeshRef - which is public API that mods hold - unchanged. +public partial class VulkanClientPlatform +{ + private bool debugDrawCalls; + + private readonly List drawCallStacks = new List(); + + [ThreadStatic] + private static FaceData[] facedataBuffer; + + public override bool DebugDrawCalls + { + get + { + return debugDrawCalls; + } + set + { + debugDrawCalls = value; + if (!value) + { + Logger.Notification("Call stacks:"); + int num = 0; + foreach (string drawCallStack in drawCallStacks) + { + Logger.Notification("{0}: {1}", num++, drawCallStack.Substring(0, 600)); + } + } + drawCallStacks.Clear(); + } + } + + public override void RenderMesh(MeshRef modelRef) + { + RuntimeStats.drawCallsCount++; + if (debugDrawCalls) + { + drawCallStacks.Add(Environment.StackTrace); + } + VAO vAO = (VAO)modelRef; + if (vAO.VaoId == 0 || vAO.Disposed) + { + if (vAO.VaoId == 0) + { + throw new ArgumentException("Fatal: Trying to render an uninitialized mesh"); + } + throw new ArgumentException("Fatal: Trying to render a disposed mesh"); + } + // Phase 3b stage 2: a draw under the vanilla gui program goes native (NativeGui.cs). + if (TryRenderGuiMeshNative(modelRef)) + { + RuntimeStats.drawCallsCount--; // DrawNativeGuiMesh counted it already + return; + } + if (TryRenderMinimalGuiNative(modelRef)) + { + RuntimeStats.drawCallsCount--; // DrawNativeGuiMesh counted it already + return; + } + if (TryRenderCloudsNative(modelRef)) + { + RuntimeStats.drawCallsCount--; // the cloud pass counted it already + return; + } + if (TryRenderStandardMeshNative(modelRef)) return; + RuntimeStats.drawCallsCount--; // the stated route counts what it records + TryDrawStated(vAO, 1, null, null, 0); + } + + public override void RenderFullscreenTriangle(MeshRef modelRef) + { + // The post passes generate their three vertices in the shader, so the + // mesh carries no buffers and none are bound. + TryDrawStated(null, 1, null, null, 0); + } + + public override void RenderMesh(MeshRef modelRef, int[] indices, int[] indicesSizes, int groupCount, bool useSSBOs) + { + RuntimeStats.drawCallsCount++; + VAO vAO = (VAO)modelRef; + // Phase 3b stage 2: inside a ChunkRenderer draw group this is a native multi-draw of + // the pool, recorded by VulkanClientPlatform.NativeChunks.cs. Outside one - the decal + // pool, a mod's pool, or with NativeChunksEnabled off - it is the generic stated draw. + if (TryDrawChunkPoolNative(vAO, indices, indicesSizes, groupCount)) return; + + // Phase 3b stage 2: inside SystemRenderDecals' BeginDecalPass/EndDecalPass scope this is + // the decal pool's multi-draw - vanilla MeshDataPool.Draw's own RenderMesh call - and + // VulkanClientPlatform.NativeWorld.cs records it as a native pass. + if (TryDrawDecalPoolNative(modelRef, indices, indicesSizes, groupCount)) return; + + // Any other pool: one indirect multi-draw from the stated state. GL takes byte offsets + // into the index buffer; the device converts them to index counts. + RuntimeStats.drawCallsCount--; + TryDrawStated(vAO, 1, indices, indicesSizes, groupCount); + } + + public override void RenderMeshInstanced(MeshRef modelRef, int quantity = 1) + { + RuntimeStats.drawCallsCount++; + VAO vAO = (VAO)modelRef; + if (TryRenderParticles2dNative(modelRef, quantity)) { RuntimeStats.drawCallsCount--; return; } + RuntimeStats.drawCallsCount--; + if (quantity > 0) TryDrawStated(vAO, quantity, null, null, 0); + } + + public override void UpdateMesh(MeshRef modelRef, MeshData data) + { + VAO vAO = (VAO)modelRef; + device.UpdateMesh(vAO.VaoId, data); + } + + /// + /// The device allocates the per-attribute buffers and derives the vertex layout from + /// which parts are present, matching the slot ordering the GL body assigns. + /// + public override MeshRef AllocateEmptyMesh(int xyzSize, int normalsSize, int uvSize, int rgbaSize, int flagsSize, int indicesSize, CustomMeshDataPartFloat customFloats, CustomMeshDataPartShort customShorts, CustomMeshDataPartByte customBytes, CustomMeshDataPartInt customInts, EnumDrawMode drawMode = EnumDrawMode.Triangles, bool staticDraw = true) + { + VAO vAO = new VAO(); + vAO.VaoId = device.CreateEmptyMesh( + xyzSize, normalsSize, uvSize, rgbaSize, flagsSize, indicesSize, + customFloats, customShorts, customBytes, customInts, + drawMode, staticDraw, ssbo: false); + vAO.IndicesCount = indicesSize; + vAO.drawMode = DrawModeToPrimiteType(drawMode); + vAO.Persistent = !staticDraw; + return vAO; + } + + /// The device builds the buffers and returns its own handle. + public override MeshRef UploadMesh(MeshData data) + { + VAO optimumUploadVao = new VAO(); + optimumUploadVao.VaoId = device.CreateMesh(data, true); + optimumUploadVao.IndicesCount = data.IndicesCount; + optimumUploadVao.drawMode = DrawModeToPrimiteType(data.mode); + return optimumUploadVao; + } + + public override void DeleteMesh(MeshRef modelref) + { + if (modelref != null) + { + // The GL body's shape: VAO.Dispose reaches DeleteVertexArrayHandles, which + // releases the device mesh (deferred until the GPU is done with the frames + // that drew it). Releasing it here as well would free the id twice. + ((VAO)modelref).Dispose(); + } + } + + /// + /// The face packing is the GL body's, unchanged; the face records go to the same + /// storage buffer, reached through the device's mesh handle instead of the buffer name. + /// + public override void UpdateSSBOMesh(MeshRef modelRef, MeshData data) + { + if (data.xyz == null) + { + return; + } + VAO vAO = (VAO)modelRef; + int verticesCount = data.VerticesCount; + if (facedataBuffer == null || facedataBuffer.Length < verticesCount / 4) + { + facedataBuffer = new FaceData[verticesCount / 4]; + } + float[] xyz = data.xyz; + float[] uv = data.Uv; + int[] flags = data.Flags; + int[] array = ((data.CustomInts != null && data.CustomInts.Count > 0) ? data.CustomInts.Values : null); + int num = ((data.CustomInts == null || data.CustomInts.Count <= 0) ? 1 : (data.CustomInts.InterleaveStride / 4)); + FaceData[] array2 = facedataBuffer; + for (int i = 0; i < verticesCount; i += 4) + { + float num2 = uv[i * 2]; + float num3 = uv[i * 2 + 1]; + float num4 = uv[i * 2 + 3]; + float num5 = uv[i * 2 + 4]; + float num6 = uv[i * 2 + 5]; + if (num2 < -1.5E-05f || num2 > 1.000015f || num3 < -1.5E-05f || num3 > 1.000015f) + { + num2 = 0f; + num3 = 0f; + } + if (num5 < -1.5E-05f || num5 > 1.000015f || num6 < -1.5E-05f || num6 > 1.000015f) + { + num5 = 0f; + num6 = 0f; + } + bool rotateUV; + if (rotateUV = num3 == num4) + { + float num7 = uv[i * 2 + 2]; + if (num5 != num7) + { + rotateUV = false; + } + } + array2[i / 4] = new FaceData(xyz, i * 3, num2, num3, num5 - num2, num6 - num3, flags, i, (array != null) ? array[i * num] : 0, rotateUV); + } + int num8 = data.XyzOffset / 12 * 16; + // Everything the mesh carries as ordinary vertex data goes first, the + // same call UpdateMesh makes. The packed face records follow, because + // they occupy the xyz slot and have to be what remains there. + device.UpdateMesh(vAO.VaoId, data); + + GCHandle optimumFacePin = GCHandle.Alloc(facedataBuffer, GCHandleType.Pinned); + try + { + device.UpdateMeshStorageBuffer(vAO.VaoId, + optimumFacePin.AddrOfPinnedObject(), num8, 16 * verticesCount); + } + finally + { + optimumFacePin.Free(); + } + } + + public override MeshRef AllocateEmptySSBOMesh(int xyzSize, int normalsSize, int uvSize, int rgbaSize, int flagsSize, int indicesSize, CustomMeshDataPartFloat customFloats, CustomMeshDataPartShort customShorts, CustomMeshDataPartByte customBytes, CustomMeshDataPartInt customInts, EnumDrawMode drawMode = EnumDrawMode.Triangles, bool staticDraw = true) + { + VAO vAO = new VAO(); + vAO.VaoId = device.CreateEmptyMesh( + xyzSize, normalsSize, uvSize, rgbaSize, flagsSize, indicesSize, + customFloats, customShorts, customBytes, customInts, + drawMode, staticDraw, ssbo: true); + vAO.IndicesCount = indicesSize; + vAO.drawMode = DrawModeToPrimiteType(drawMode); + vAO.Persistent = !staticDraw; + return vAO; + } + + /// ClientPlatformWindows.DrawModeToPrimiteType, which is private. + private static PrimitiveType DrawModeToPrimiteType(EnumDrawMode drawmode) + { + return (PrimitiveType)(drawmode switch + { + EnumDrawMode.Lines => 1, + EnumDrawMode.LineStrip => 3, + _ => 4, + }); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.ModPasses.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.ModPasses.cs new file mode 100644 index 00000000..6c0ee5c3 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.ModPasses.cs @@ -0,0 +1,295 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 5: the mod pass API hosted on the frame graph. A mod registers an +// OptimumPassDecl (slot, reads and writes by well-known handle, draw callback, optional motion +// writer) through OptimumModPasses; at the end of each render stage's bracket, after the stage's +// RegisterRenderer renderers, the platform runs the passes declared for that slot: it binds the +// declared target, declares the pass with the declared colour slots and reads (mod-hosted, so +// OpenSampling and AllowSplit), opens the motion window when the pass is a motion writer, calls +// the draw, ends the pass and restores the target and pass context. A RegisterRenderer renderer +// opens a registered writer's window itself through OptimumModPasses.BeginMotionWriter, which +// reaches the hooks installed here. OpenGL (ClientPlatformWindows) reads none of it. +public partial class VulkanClientPlatform +{ + /// The flags of every mod-hosted pass (plan, section D and Phase 2 step 2). + internal const PassFlags ModPassFlags = PassFlags.OpenSampling | PassFlags.AllowSplit; + + /// Mod passes that ran their draw. + internal long ModPassesRun { get; private set; } + + /// Mod passes skipped because a written attachment does not exist this session. + internal long ModPassesSkipped { get; private set; } + + /// The pass whose draw is running, null outside one ("Mod/<mod>/<name>/<target>"). + internal string? CurrentModPass { get; private set; } + + private readonly Dictionary modPassPlans = new(ReferenceEqualityComparer.Instance); + private readonly HashSet modPassFailuresLogged = new(ReferenceEqualityComparer.Instance); + private List? modPassPlansFor; + private int modPassPlansMotionIndex = int.MinValue; + + /// A registration resolved against the current framebuffer set; rebuilt when the set changes. + private sealed class ModPassPlan + { + public FrameBufferRef? Target; + public bool Runnable; + public string SkipReason = ""; + public PassDeclaration Declaration = new(); + } + + private void InstallModPassHooks() + { + OptimumModPasses.MotionBeginHook = BeginModMotionWriter; + OptimumModPasses.MotionEndHook = EndModMotionWriter; + } + + private void RemoveModPassHooks() + { + if (OptimumModPasses.MotionBeginHook?.Target == this) OptimumModPasses.MotionBeginHook = null; + if (OptimumModPasses.MotionEndHook?.Target == this) OptimumModPasses.MotionEndHook = null; + modPassPlans.Clear(); + modPassPlansFor = null; + } + + /// A renderer's registered writer: the window opens only inside Opaque or AfterOIT. + private bool BeginModMotionWriter(OptimumMotionWriterDecl writer) + { + if (device == null || writer == null || !InRenderStage) return false; + if (!OptimumPassContract.IsMotionWindowSlot((EnumOptimumPass)(int)CurrentRenderStage)) return false; + return writer.Mode == EnumOptimumMotionWrite.MotionOnly ? BeginMotionOnlyWrite() : BeginMotionWrite(); + } + + private void EndModMotionWriter() + { + if (device == null) return; + EndMotionWrite(); + } + + /// Runs the passes declared for , in registration order. + internal void RunModPasses(EnumRenderStage stage) + { + if (device == null) return; + OptimumPassRegistration[] passes = OptimumModPasses.ForSlot((EnumOptimumPass)(int)stage); + if (passes.Length == 0) return; + + FrameBufferRef saved = CurrentFrameBuffer; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + foreach (OptimumPassRegistration registration in passes) + { + RunModPass(registration); + } + passContext = outer; + passContextFlags = outerFlags; + CurrentFrameBuffer = saved; + } + + private void RunModPass(OptimumPassRegistration registration) + { + OptimumPassDecl decl = registration.Decl; + ModPassPlan plan = PlanFor(registration); + if (!plan.Runnable) + { + ModPassesSkipped++; + if (modPassFailuresLogged.Add(decl)) + Logger?.Warning("[Optimum] mod pass '{0}' of {1} skipped: {2}", decl.Name, registration.ModId, plan.SkipReason); + return; + } + + passContext = "Mod/" + registration.ModId + "/" + decl.Name; + passContextFlags = ModPassFlags; + // The declared slots are the draw buffers the mod's draws write, for this pass only; every + // draw inside it is recorded under the declaration (name, slots, reads, flags) on the + // plan's target by the generic stated route. + int targetId = plan.Target?.FboId ?? PassDeclaration.DefaultFramebuffer; + uint savedDrawBuffers = stated.DrawBuffers(targetId); + CurrentFrameBuffer = plan.Target!; + stated.SetDrawBuffers(targetId, plan.Declaration.ColorSlots); + + bool motion = false; + CurrentModPass = plan.Declaration.Name; + statedPass = plan.Declaration; + try + { + if (decl.MotionWriter != null) + { + motion = decl.MotionWriter.Mode == EnumOptimumMotionWrite.MotionOnly ? BeginMotionOnlyWrite() : BeginMotionWrite(); + } + decl.Draw(decl); + ModPassesRun++; + } + catch (Exception error) + { + // A mod's draw must not take the frame down with it; the pass is reported once. + if (modPassFailuresLogged.Add(decl)) + Logger?.Error("[Optimum] mod pass '{0}' of {1} threw: {2}", decl.Name, registration.ModId, error); + } + finally + { + if (motion) EndMotionWrite(); + CurrentModPass = null; + statedPass = null; + stated.SetDrawBuffers(targetId, savedDrawBuffers); + } + } + + private ModPassPlan PlanFor(OptimumPassRegistration registration) + { + List buffers = FrameBuffers; + if (!ReferenceEquals(buffers, modPassPlansFor) || modPassPlansMotionIndex != MotionAttachmentIndex) + { + modPassPlans.Clear(); + modPassFailuresLogged.Clear(); + modPassPlansFor = buffers; + modPassPlansMotionIndex = MotionAttachmentIndex; + } + if (!modPassPlans.TryGetValue(registration.Decl, out ModPassPlan? plan)) + { + plan = BuildModPassPlan(registration); + modPassPlans[registration.Decl] = plan; + } + return plan; + } + + /// Resolves the declared handles to the target, its colour slots and the read textures. + private ModPassPlan BuildModPassPlan(OptimumPassRegistration registration) + { + OptimumPassDecl decl = registration.Decl; + var plan = new ModPassPlan(); + EnumOptimumTarget target = OptimumPassContract.TargetOf(decl); + int targetIndex = target switch + { + EnumOptimumTarget.Primary => PrimaryIndex, + EnumOptimumTarget.Transparent => TransparentIndex, + _ => -1, + }; + + string targetName; + uint colorSlots = 0; + if (target == EnumOptimumTarget.Default) + { + plan.Target = null; + targetName = "Default"; + colorSlots = uint.MaxValue; + } + else + { + List buffers = FrameBuffers; + FrameBufferRef? buffer = buffers != null && targetIndex >= 0 && targetIndex < buffers.Count ? buffers[targetIndex] : null; + if (buffer == null) + { + plan.SkipReason = target + " does not exist"; + return plan; + } + plan.Target = buffer; + targetName = targetIndex.ToString(CultureInfo.InvariantCulture); + foreach (EnumOptimumAttachment write in decl.Writes) + { + if (write == EnumOptimumAttachment.PrimaryDepth) + { + if (buffer.DepthTextureId <= 0) + { + plan.SkipReason = "PrimaryDepth does not exist"; + return plan; + } + continue; + } + int slot = OptimumPassContract.ColorSlotOf(write); + if (slot < 0 || buffer.ColorTextureIds == null || slot >= buffer.ColorTextureIds.Length || + buffer.ColorTextureIds[slot] <= 0 || (target == EnumOptimumTarget.Primary && slot == MotionAttachmentIndex)) + { + plan.SkipReason = write + " does not exist this session"; + return plan; + } + colorSlots |= 1u << slot; + } + if (decl.MotionWriter != null) + { + // The window writes the motion attachment on top of the declared slots; without one + // the begin call refuses and the draw falls back to camera reprojection. + // Undeclared colour slots stay out of the scope even though the window's draw-buffer + // mask names them, so a pass can still sample them (the final composition's shape). + if (MotionAttachmentIndex > -1) colorSlots |= 1u << MotionAttachmentIndex; + } + } + + var reads = new List(); + foreach (EnumOptimumAttachment read in decl.Reads) + { + int id = TextureOf(read); + if (id > 0 && !reads.Contains(id)) reads.Add(id); + } + + plan.Declaration = new PassDeclaration + { + Name = "Mod/" + registration.ModId + "/" + decl.Name + "/" + targetName, + FramebufferId = target == EnumOptimumTarget.Default + ? PassDeclaration.DefaultFramebuffer + : plan.Target!.FboId, + ColorSlots = colorSlots, + Reads = reads.ToArray(), + Flags = ModPassFlags, + }; + plan.Runnable = true; + return plan; + } + + /// The texture behind a handle in the current framebuffer set, 0 when it does not exist. + internal int TextureOf(EnumOptimumAttachment attachment) + { + switch (attachment) + { + case EnumOptimumAttachment.PrimaryMotion: + return MotionAttachmentIndex >= 0 ? ColourOf(PrimaryIndex, MotionAttachmentIndex) : 0; + case EnumOptimumAttachment.PrimaryDepth: + return DepthOf(PrimaryIndex); + case EnumOptimumAttachment.LiquidDepth: + return DepthOf(LiquidDepthIndex); + case EnumOptimumAttachment.ShadowFarDepth: + return DepthOf(ShadowFarIndex); + case EnumOptimumAttachment.ShadowNearDepth: + return DepthOf(ShadowNearIndex); + case EnumOptimumAttachment.GodRays: + return ColourOf(GodRaysIndex, 0); + case EnumOptimumAttachment.BloomLowRes: + return ColourOf(BlurVerticalLowResIndex, 0); + case EnumOptimumAttachment.Luma: + return ColourOf(LumaIndex, 0); + case EnumOptimumAttachment.SsaoBlurred: + return ColourOf(SsaoBlurVerticalIndex, 0); + } + EnumOptimumTarget target = OptimumPassContract.TargetOf(attachment); + int slot = OptimumPassContract.ColorSlotOf(attachment); + if (slot < 0) return 0; + if (target == EnumOptimumTarget.Primary) + { + // Absent G-buffer: slot 2 is the motion attachment, which is not this handle. + if (slot == MotionAttachmentIndex) return 0; + return ColourOf(PrimaryIndex, slot); + } + return target == EnumOptimumTarget.Transparent ? ColourOf(TransparentIndex, slot) : 0; + } + + private int ColourOf(int index, int slot) + { + List buffers = FrameBuffers; + if (buffers == null || index < 0 || index >= buffers.Count) return 0; + FrameBufferRef buffer = buffers[index]; + if (buffer?.ColorTextureIds == null || slot < 0 || slot >= buffer.ColorTextureIds.Length) return 0; + return buffer.ColorTextureIds[slot]; + } + + private int DepthOf(int index) + { + List buffers = FrameBuffers; + if (buffers == null || index < 0 || index >= buffers.Count || buffers[index] == null) return 0; + return buffers[index].DepthTextureId; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeBlit.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeBlit.cs new file mode 100644 index 00000000..a456f3ef --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeBlit.cs @@ -0,0 +1,294 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), stage 1: the blit to +// the swapchain-equivalent Default target runs natively. The three branches are the GL body's +// (ClientPlatformWindows.BlitPrimaryToDefault): the TAA debug view, FSR (EASU into the FSR +// target, RCAS into Default) and the plain blit, with the same conditions and uniform values. +// EASU reads Primary colour 0 after late overlays (decision 7); the plain blit reads slot 21 when +// the post-composition TAA sharpen ran. Each written target +// is one declared pass, and no draw between BeginNativePass and EndNativePass touches the GL +// state tracker, a texture unit or a draw-buffer mask. +public partial class VulkanClientPlatform +{ + /// + /// False runs the OpenGL body on the Vulkan device instead of the native chain: the old + /// route the parity tests compare the native one against. + /// + internal bool NativeBlitEnabled { get; set; } = true; + + /// + /// The base's OffscreenBuffer, which is private there. Tracked through the one virtual + /// that changes it, and starting where the base's own field does. + /// + private bool offscreenBufferActive = true; + + public override void ToggleOffscreenBuffer(bool enable) + { + offscreenBufferActive = enable; + base.ToggleOffscreenBuffer(enable); + } + + /// A native fullscreen program: its pipeline for the current target, and the placements its draws write through. + private sealed class NativeFullscreenPass + { + public NativeFullscreenPass(string passName, string[] uniforms, string[] samplers) + { + PassName = passName; + UniformNames = uniforms; + SamplerNames = samplers; + Uniforms = new NativeUniform[uniforms.Length]; + Samplers = new NativeSamplerSlot[samplers.Length]; + } + + public string PassName { get; } + public string[] UniformNames { get; } + public string[] SamplerNames { get; } + + /// Resolved once per pipeline, never by name per draw. + public NativeUniform[] Uniforms; + public NativeSamplerSlot[] Samplers; + + public NativePipeline? Pipeline; + public RenderTargetFormats? Formats; + public bool Reported; + + public void Adopt(NativePipeline pipeline, RenderTargetFormats formats) + { + Pipeline = pipeline; + Formats = formats; + for (int i = 0; i < UniformNames.Length; i++) Uniforms[i] = pipeline.Uniform(UniformNames[i]); + for (int i = 0; i < SamplerNames.Length; i++) Samplers[i] = pipeline.Sampler(SamplerNames[i]); + } + } + + private readonly NativeFullscreenPass nativeTaaDebug = + new("taa-debug", new[] { "mode", "renderSize" }, new[] { "motionTex", "depthTex", "sceneTex" }); + + private readonly NativeFullscreenPass nativeFsrEasu = + new("fsr-easu", new[] { "inputTexelSize" }, new[] { "inputScene" }); + + private readonly NativeFullscreenPass nativeFsrRcas = + new("fsr-rcas", new[] { "inputTexelSize" }, new[] { "inputScene" }); + + private readonly NativeFullscreenPass nativeBlit = + new("blit", Array.Empty(), new[] { "scene" }); + + /// Opaque fullscreen state: no blend, no depth, no culling, every channel written. + private static AttachmentBlend[] OpaqueColorZero() => new[] { AttachmentBlend.Default }; + + /// + /// The pipeline for one fullscreen program against one target, rebuilt only when the + /// program was relinked or the target's formats changed. Opaque unless the pass states a + /// blend; a pass object always states the same one, so the blend is not part of the check. + /// + private NativePipeline? NativePipelineFor(NativeFullscreenPass pass, ShaderProgramBase program, int framebufferId, + AttachmentBlend[]? blend = null) + { + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, 1u); + if (formats == null) return null; + + if (pass.Pipeline != null && pass.Pipeline.ProgramId == program.ProgramId && + formats.Equals(pass.Formats) && device.IsNativePipelineLive(pass.Pipeline)) + { + return pass.Pipeline; + } + + NativePipeline? pipeline = device.RequestNativePipeline(new NativePipelineDescription + { + ProgramId = program.ProgramId, + PassName = pass.PassName, + Blend = blend ?? OpaqueColorZero(), + DepthTest = false, + DepthWrite = false, + Cull = CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + Targets = formats, + }, out string error); + + if (pipeline == null) + { + if (!pass.Reported) + { + pass.Reported = true; + Logger.Warning("Optimum: no native pipeline for '{0}': {1}", pass.PassName, error); + } + pass.Pipeline = null; + return null; + } + + pass.Reported = false; + pass.Adopt(pipeline, formats); + return pipeline; + } + + /// The Default target, as a native pass names it. + private const int NativeDefaultTarget = PassDeclaration.DefaultFramebuffer; + + private bool BeginNativeBlitPass(string name, int framebufferId, int width, int height, int[] reads) => + device.BeginNativePass(new NativePassDescription + { + Name = name, + FramebufferId = framebufferId, + ColorSlots = 1u, + Reads = reads, + Flags = PassFlags.None, + ViewportWidth = width, + ViewportHeight = height, + }); + + /// + /// Leaves the stated state where the OpenGL body leaves it, so everything the + /// client draws after the blit (the ortho GUI pass) sees what it always saw: the Default + /// target bound, the viewport on the window, and blending back on where the body turned + /// it off. Outside every native pass. + /// + private void FinishNativeBlit(bool restoreBlend) + { + passContext = "Frame"; + passContextFlags = PassFlags.AllowSplit; + LoadFrameBuffer(EnumFrameBuffer.Default); + if (restoreBlend) GlToggleBlend(true); + } + + /// The native blit: the OpenGL body's three branches, drawn through the native device API. + private void RenderNativeBlit() + { + if (!offscreenBufferActive) return; + + List buffers = FrameBuffers; + FrameBufferRef primary = buffers[0]; + int scene2D = primary.ColorTextureIds[0]; + Size2i client = OptimumWindowClientSize(); + + // TAA debug views (P1): bypasses FSR and the blit entirely, exactly as the GL body does. + if (OptimumConfig.TaaDebugView != 0 && MotionAttachmentIndex >= 0) + { + ShaderProgram taaDebug = ShaderPrograms.TaaDebug; + if (taaDebug != null && !taaDebug.LoadError) + { + int motion = primary.ColorTextureIds[MotionAttachmentIndex]; + int depth = primary.DepthTextureId; + NativePipeline? pipeline = NativePipelineFor(nativeTaaDebug, taaDebug, NativeDefaultTarget); + if (pipeline != null && + BeginNativeBlitPass("Blit/Default", NativeDefaultTarget, client.Width, client.Height, + new[] { motion, depth, scene2D })) + { + device.WriteNative(pipeline, nativeTaaDebug.Uniforms[0], OptimumConfig.TaaDebugView); + device.WriteNative(pipeline, nativeTaaDebug.Uniforms[1], primary.Width, primary.Height); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeTaaDebug.Samplers[0], motion), + new NativeTexture(nativeTaaDebug.Samplers[1], depth), + new NativeTexture(nativeTaaDebug.Samplers[2], scene2D), + }); + } + device.EndNativePass(); + NotePostStep(NativePostStep.Blit); + FinishNativeBlit(restoreBlend: false); + return; + } + } + + // AfterFinalComposition renderers draw onto Primary between FinalComposition and this + // method. Sharpen only now, after those overlays are complete; FSR owns RCAS instead. + bool useFsr = OptimumFsrBlitActive(); + if (!useFsr) + { + scene2D = RenderOptimumTaaSharpen(scene2D); + } + NotePostStep(NativePostStep.Blit); + if (useFsr) + { + FrameBufferRef fsrTarget = buffers[OptimumFsrFramebufferIndex]; + try + { + ShaderProgram fsrEasu = ShaderPrograms.FsrEasu; + ShaderProgram fsrRcas = ShaderPrograms.FsrRcas; + int fsrColor = fsrTarget.ColorTextureIds[0]; + + // EASU upsamples Primary colour 0 into the FSR target (decision 7). + NativePipeline? easu = NativePipelineFor(nativeFsrEasu, fsrEasu, fsrTarget.FboId); + if (easu != null && + BeginNativeBlitPass("Blit/" + OptimumFsrFramebufferIndex, fsrTarget.FboId, + fsrTarget.Width, fsrTarget.Height, new[] { scene2D })) + { + device.WriteNative(easu, nativeFsrEasu.Uniforms[0], 1f / primary.Width, 1f / primary.Height); + device.DrawNativeFullscreen(easu, new[] + { + new NativeTexture(nativeFsrEasu.Samplers[0], scene2D), + }); + } + device.EndNativePass(); + + // RCAS sharpens the upsampled image into Default. + NativePipeline? rcas = NativePipelineFor(nativeFsrRcas, fsrRcas, NativeDefaultTarget); + if (rcas != null && + BeginNativeBlitPass("Blit/Default", NativeDefaultTarget, client.Width, client.Height, new[] { fsrColor })) + { + device.WriteNative(rcas, nativeFsrRcas.Uniforms[0], 1f / fsrTarget.Width, 1f / fsrTarget.Height); + device.DrawNativeFullscreen(rcas, new[] + { + new NativeTexture(nativeFsrRcas.Samplers[0], fsrColor), + }); + } + device.EndNativePass(); + FinishNativeBlit(restoreBlend: true); + return; + } + catch (Exception error) + { + device.EndNativePass(); + DisableOptimumFsr(error); + FinishNativeBlit(restoreBlend: true); + } + } + + // The TAA sharpen is deliberately after final composition and its late overlays. A plain + // blit therefore reads its dedicated target, while FSR above keeps Primary as input and + // supplies the one RCAS pass at native resolution. + int finalScene = NativeFinalBlitSceneTexture(scene2D, useFsr); + ShaderProgramBlit blit = ShaderPrograms.Blit; + NativePipeline? plain = NativePipelineFor(nativeBlit, blit, NativeDefaultTarget); + if (plain != null && + BeginNativeBlitPass("Blit/Default", NativeDefaultTarget, client.Width, client.Height, new[] { finalScene })) + { + device.DrawNativeFullscreen(plain, new[] + { + new NativeTexture(nativeBlit.Samplers[0], finalScene), + }); + } + device.EndNativePass(); + FinishNativeBlit(restoreBlend: false); + } + + /// + /// Returns the post-composition TAA sharpen target for the plain blit, when the same guards + /// that ran prove that this frame produced it. FSR is + /// checked first: its EASU/RCAS branch owns the final sharpening and must consume Primary's + /// unsharpened composition. + /// + private int NativeFinalBlitSceneTexture(int fallback, bool fsrActive) + { + ShaderProgram sharpen = ShaderPrograms.TaaSharpen; + List buffers = FrameBuffers; + FrameBufferRef target = buffers != null && buffers.Count > OptimumTaaSharpenIndex + ? buffers[OptimumTaaSharpenIndex] + : null!; + bool targetAvailable = sharpen != null && !sharpen.LoadError && target != null && + !target.Disposed && target.ColorTextureIds != null && target.ColorTextureIds.Length > 0; + int sharpened = targetAvailable ? target.ColorTextureIds[0] : fallback; + return OptimumApiBridge.SelectTaaPresentationTexture( + fallback, sharpened, fsrActive, TaaResolvedThisFrame, OptimumConfig.TaaSharpness, + targetAvailable); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeChunks.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeChunks.cs new file mode 100644 index 00000000..80d24e65 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeChunks.cs @@ -0,0 +1,486 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b decision 5 +// stage 2: the terrain, the heaviest draw path in the game, on the native device API. +// +// What it draws: every ChunkRenderer draw group - the two shadow cascades, the opaque, +// topsoil, vegetation, blend-no-cull and decorative groups of the Opaque stage, the OIT +// liquid and transparent groups, the liquid velocity redraw and the AfterOIT terrain +// overlay. Each group is one indirect multi-draw per visible chunk pool over per-chunk +// meshes, with the FaceData storage buffer feeding the vertex fetch when the pool is an +// SSBO pool, and it stays exactly that: MeshDataPool's ranges reach +// VulkanDevice.DrawNativeMeshMulti, which allocates from the same per-slot indirect ring +// the emulated DrawMeshMulti allocates from, and the real mesh id reaches BindProgramSets +// so the storage-buffer vertex fetch resolves per draw. +// Where the other side is: ClientPlatformAbstract.BeginChunkPass / EndChunkPass have +// neutral bodies (false and nothing), so the OpenGL path still draws through +// ClientPlatformWindows.RenderMesh(MeshRef, int[], int[], int, bool) -> GL.MultiDrawElements +// under the GlToggleBlend / GlEnableDepthTest / GlDepthMask / cull calls ChunkRenderer +// already makes. NativeChunksEnabled false takes that same route on the Vulkan device, +// which is what the differential test compares against. +// Target and slots: whatever the stage bound. Primary for the opaque, overlay and liquid +// velocity groups; the Transparent target for the OIT groups; a shadow map for the +// cascades. The motion window is the pass's colour-write mask and never a draw-buffer +// toggle: the motion attachment joins the pass's slots while the window is open, with +// replace blending, and the liquid velocity group masks every other slot to zero, which is +// what BeginMotionOnlyWrite means here. +// State that is not obvious: +// - the per-attachment blend of a Primary group is the client's own contract, rebuilt +// from the same values GlToggleBlend applies: the standard mode on the shaded slots, +// replace blending on the SSAO G-buffer slots 2 and 3 and on the motion attachment, +// because a blended G-buffer or motion value is an average of two surfaces and belongs +// to neither; +// - the Transparent target's contract is whichever set the client last applied to it - +// the vanilla three-attachment set or the six-attachment OIT accumulation set - +// recorded where the client states it (ApplyTransparentPassBlendState and +// BeginOitAccumulation), not read back out of the state tracker; +// - chunkliquid samples the depth attachment it draws against with depth writes off, so +// its pipeline declares SamplesBoundDepth and the scope holds that depth read-only; +// - the slots the group's program does not write keep their contents, because the +// pipeline masks every output the program never writes (rule 9). +// What is still emulated inside a chunk group, and why: the uniform values the group sets +// by name - the per-pool "origin" from MeshDataPoolManager and the generated program +// setters - still travel through the GL-shaped uniform dispatch. They land in the same +// per-program record and push shadows the native draw snapshots, so the image is the same; +// removing that dispatch means moving MeshDataPoolManager itself, which is API-fork code +// and a later stage. The draws, the pass, the pipelines, the fixed state and the texture +// resolution are all native. +// What pins it: NativeChunkTests (old route against native route, per pass, including the +// motion attachment) and Optimum.Tests/native-world-systems-coverage-tests.cs. +public partial class VulkanClientPlatform +{ + /// + /// False runs the chunk groups through the generic stated multi-draw instead of the native + /// pass: the route the differential test compares against, in the pattern of + /// . + /// + internal bool NativeChunksEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_CHUNKS") != "0"; + + /// + /// The texture each program sampler was last pointed at, recorded where the client points + /// it (). A native draw resolves what it samples from + /// handles, so it needs the handle the client chose rather than the texture unit the + /// GL-shaped path bound it to. Keyed by program id, then by the sampler's name. + /// + private readonly Dictionary> nativeProgramTextures = new(); + + /// The sampler names of a program, resolved once from its interface rather than per draw. + private readonly Dictionary nativeProgramSamplers = new(); + + /// + /// The per-attachment blend contract the Transparent target is drawn under, recorded at + /// the two seams the client states it through. Null until the client states one. + /// + private AttachmentBlend[]? nativeTransparentBlend; + + /// + /// The colour slots the client last selected on the Transparent target, recorded with the + /// blend contract: 0x7 for the vanilla set (ApplyTransparentPassBlendState), 0x3F for the OIT + /// accumulation set (BeginOitAccumulation), whose colour accumulation lives on slots 3-5 - + /// attachments the target's FrameBufferRef does not list, so its texture count is not the + /// slot set. 0 until the client has stated one. + /// + private uint nativeTransparentSlots; + + /// Bumped when that contract changes, so its pipelines are rebuilt rather than reused. + private int nativeBlendEpoch; + + // ------------------------------------------------------------------ the open scope + + private bool chunkScopeActive; + private bool chunkScopeBlend; + private bool chunkScopeDepthTest; + private bool chunkScopeDepthWrite; + private bool chunkScopeCull; + private bool chunkScopeMotionOnly; + private string chunkScopeName = ""; + private FrameBufferRef? chunkScopeTarget; + private uint chunkScopeSlots; + private bool chunkScopePassOpen; + private string chunkScopeOuterContext = ""; + private PassFlags chunkScopeOuterFlags; + private NativeTexture[] chunkScopeTextures = Array.Empty(); + private int[] chunkScopeReads = Array.Empty(); + + /// The native pipelines the chunk groups draw through, one per distinct shape. + private readonly Dictionary chunkPipelines = new(); + + /// + /// Every dimension of a chunk group's pipeline the platform decides: the program, the + /// target and the slots it writes, the mesh's vertex layout, and the fixed state the seam + /// stated. The device keys its own cache on the full description; this one only keeps the + /// description and its blend array from being rebuilt per draw. + /// + private readonly record struct ChunkPipelineKey(int ProgramId, int FramebufferId, uint Slots, int LayoutId, + bool Blend, bool DepthTest, bool DepthWrite, bool Cull, bool MotionOnly, bool SamplesBoundDepth, + int BlendEpoch); + + // ------------------------------------------------------------------- captured state + + /// + /// Records the texture a program sampler points at. Called from + /// and , which is + /// where the client states it; the GL-shaped unit binding still happens there too, so the + /// old route keeps working unchanged. + /// + internal void NoteNativeProgramTexture(int programId, string samplerName, int textureId) + { + if (!nativeProgramTextures.TryGetValue(programId, out Dictionary? textures)) + { + textures = new Dictionary(StringComparer.Ordinal); + nativeProgramTextures[programId] = textures; + } + textures[samplerName] = textureId; + } + + /// + /// Records one attachment of the Transparent target's blend contract, at the seam the + /// client states it through. GL's blend enable is global, so the recorded entry carries + /// the functions and the group's own blend flag decides whether they apply. + /// + internal void NoteNativeTransparentBlend(int slot, int glEquation, int srcColor, int dstColor, + int srcAlpha, int dstAlpha) + { + if ((uint)slot >= RenderLimits.MaxColorAttachments) return; + if (nativeTransparentBlend == null) + { + nativeTransparentBlend = new AttachmentBlend[RenderLimits.MaxColorAttachments]; + for (int i = 0; i < nativeTransparentBlend.Length; i++) + { + nativeTransparentBlend[i] = AttachmentBlend.Default; + } + } + + AttachmentBlend blend = AttachmentBlend.Default; + blend.Enabled = true; + blend.ColorOp = GlEnums.BlendOpFrom(glEquation); + blend.AlphaOp = blend.ColorOp; + blend.SrcColor = GlEnums.BlendFactorFrom(srcColor); + blend.DstColor = GlEnums.BlendFactorFrom(dstColor); + blend.SrcAlpha = GlEnums.BlendFactorFrom(srcAlpha); + blend.DstAlpha = GlEnums.BlendFactorFrom(dstAlpha); + if (!nativeTransparentBlend[slot].Equals(blend)) nativeBlendEpoch++; + nativeTransparentBlend[slot] = blend; + } + + // ------------------------------------------------------------------------ the seam + + /// + /// Opens the scope one ChunkRenderer draw group draws under. The group's multi-draws then + /// reach from the mesh seam; the native pass itself + /// opens with the first draw, because that is when the textures it reads are known. + /// + public override bool BeginChunkPass(string chunkPass, bool blend, bool depthTest, bool depthWrite, bool cullFace) + { + EndChunkPass(); + if (!NativeChunksEnabled || device == null) return false; + + FrameBufferRef target = CurrentFrameBuffer; + if (target == null || target.FboId <= 0) return false; + + chunkScopeActive = true; + chunkScopeName = chunkPass ?? "chunk"; + chunkScopeBlend = blend; + chunkScopeDepthTest = depthTest; + chunkScopeDepthWrite = depthWrite; + chunkScopeCull = cullFace; + chunkScopeTarget = target; + // The liquid velocity redraw is the one chunk group that opens its window with + // BeginMotionOnlyWrite, so the seam's own name is what says the other slots are masked + // off - no second copy of the window's state, and no reading a draw-buffer mask back. + chunkScopeMotionOnly = string.Equals(chunkScopeName, "chunk-liquid-motion", StringComparison.Ordinal); + chunkScopeSlots = ChunkColorSlots(target); + chunkScopePassOpen = false; + return true; + } + + /// Closes the scope and, if a draw opened one, the native pass with it. + public override void EndChunkPass() + { + if (!chunkScopeActive) return; + chunkScopeActive = false; + + if (chunkScopePassOpen) + { + chunkScopePassOpen = false; + device.EndNativePass(); + SetPassContext(chunkScopeOuterContext, chunkScopeOuterFlags); + } + chunkScopeTarget = null; + } + + /// + /// One chunk pool's multi-draw, recorded natively. False means the group is not in a native + /// scope, or the first draw of one could not be recorded, and the caller takes the generic + /// stated route. Once the scope's pass is open the native route owns the group: a draw the + /// device skips (a pipeline still compiling) is skipped, not moved to another route. + /// + internal bool TryDrawChunkPoolNative(VAO vao, int[] indicesStarts, int[] indicesSizes, int groupCount) + { + if (!chunkScopeActive || device == null) return false; + if (vao == null || vao.VaoId == 0 || vao.Disposed) return false; + if (groupCount <= 0) return chunkScopePassOpen; + + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (program == null || program.ProgramId <= 0) return chunkScopePassOpen; + + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) return chunkScopePassOpen; + + FrameBufferRef target = chunkScopeTarget!; + string[] names = ChunkSamplerNames(program.ProgramId); + int textureCount = CollectChunkTextures(program, names, target, out bool samplesBoundDepth); + NativePipeline? pipeline = ChunkPipeline(program, target, layoutId, samplesBoundDepth); + if (pipeline == null) return chunkScopePassOpen; + + for (int i = 0; i < textureCount; i++) + { + chunkScopeTextures[i] = chunkScopeTextures[i] with { Sampler = pipeline.Sampler(names[i]) }; + } + + if (!chunkScopePassOpen && !OpenChunkPass(target, textureCount)) return false; + + device.DrawNativeMeshMulti(pipeline, vao.VaoId, indicesStarts, indicesSizes, groupCount, + new ReadOnlySpan(chunkScopeTextures, 0, textureCount)); + return true; + } + + /// + /// Declares the group's pass: the bound target, the slots the group writes and the textures + /// its first draw reads, in the viewport the stage left - the way the OpenGL body's + /// bind-only setter leaves it. + /// + private bool OpenChunkPass(FrameBufferRef target, int textureCount) + { + Rect2D viewport = StatedViewport(); + chunkScopeOuterContext = passContext; + chunkScopeOuterFlags = passContextFlags; + + var reads = new int[textureCount]; + Array.Copy(chunkScopeReads, reads, textureCount); + if (!device.BeginNativePass(new NativePassDescription + { + Name = chunkScopeName + "/" + target.FboId, + FramebufferId = target.FboId, + ColorSlots = chunkScopeSlots, + Reads = reads, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + })) + { + return false; + } + chunkScopePassOpen = true; + return true; + } + + // ------------------------------------------------------------------- the pipeline + + /// + /// The pipeline for one group's program, target, mesh shape and fixed state. The device + /// caches the pipeline itself; this table only keeps the description and its blend array + /// from being rebuilt per draw. + /// + private NativePipeline? ChunkPipeline(ShaderProgramBase program, FrameBufferRef target, int layoutId, + bool samplesBoundDepth) + { + RenderTargetFormats? formats = device.NativeTargetFormats(target.FboId, chunkScopeSlots); + if (formats == null) return null; + + var key = new ChunkPipelineKey(program.ProgramId, target.FboId, chunkScopeSlots, layoutId, + chunkScopeBlend, chunkScopeDepthTest, chunkScopeDepthWrite, chunkScopeCull, + chunkScopeMotionOnly, samplesBoundDepth, nativeBlendEpoch); + if (chunkPipelines.TryGetValue(key, out NativePipeline? cached) && + device.IsNativePipelineLive(cached) && formats.Equals(cached.Description.Targets)) + { + return cached; + } + + NativePipeline? pipeline = device.RequestNativePipeline(new NativePipelineDescription + { + ProgramId = program.ProgramId, + PassName = program.PassName, + Blend = ChunkBlend(target, formats.ColorFormats.Length), + DepthTest = chunkScopeDepthTest, + DepthWrite = chunkScopeDepthWrite, + DepthCompare = CompareOp.Less, + Cull = chunkScopeCull ? CullModeFlags.BackBit : CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + VertexLayoutId = layoutId, + SamplesBoundDepth = samplesBoundDepth, + Targets = formats, + }, out string error); + + if (pipeline == null) + { + chunkPipelines.Remove(key); + if (RenderTrace.Enabled) + { + RenderTrace.Write("no native chunk pipeline for '" + chunkScopeName + "': " + error); + } + return null; + } + chunkPipelines[key] = pipeline; + return pipeline; + } + + /// + /// The colour slots a chunk group's pass writes. On Primary that is the client's default + /// set, plus the motion attachment while a motion window is open; on any other target it is + /// every bound slot, which is the set the emulated draw would have written. + /// + private uint ChunkColorSlots(FrameBufferRef target) + { + if (IsTransparentTarget(target) && nativeTransparentSlots != 0) return nativeTransparentSlots; + if (!IsPrimaryTarget(target)) return NativeAllColorSlots(target); + + int motion = MotionAttachmentIndex; + if (motion >= 0 && OptimumMotionWriteActive) return (1u << (motion + 1)) - 1u; + return motion > 0 ? (1u << motion) - 1u : NativeWorldColorSlots(); + } + + /// + /// The per-attachment blend of the group, rebuilt from the values the client applied + /// rather than read back off the state tracker. + /// + private AttachmentBlend[] ChunkBlend(FrameBufferRef target, int count) + { + var blend = new AttachmentBlend[Math.Max(count, 1)]; + bool primary = IsPrimaryTarget(target); + bool transparent = !primary && nativeTransparentBlend != null && IsTransparentTarget(target); + int motion = primary && OptimumMotionWriteActive ? MotionAttachmentIndex : -1; + + for (int slot = 0; slot < blend.Length; slot++) + { + AttachmentBlend entry; + if (transparent) + { + // The contract the client applied to the Transparent target, whichever of the + // two it was; its enable is the group's, because GL's is global. + entry = nativeTransparentBlend![slot]; + entry.Enabled = chunkScopeBlend; + } + else if (chunkScopeBlend && primary && OptimumRenderSsao && (slot == 2 || slot == 3)) + { + // GlToggleBlend's own exception: the SSAO G-buffer is replace-blended, because + // a blended normal or position belongs to neither surface. + entry = ReplaceBlend(chunkScopeBlend); + } + else + { + entry = AttachmentBlend.Default; + entry.Enabled = chunkScopeBlend; + } + + // The motion attachment never blends (ApplyOptimumMotionBlendState). + if (slot == motion) entry = ReplaceBlend(chunkScopeBlend); + + // The liquid velocity redraw writes the motion attachment and nothing else: the + // window is this mask, not a draw-buffer toggle. + if (chunkScopeMotionOnly && slot != motion) entry.WriteMask = 0; + blend[slot] = entry; + } + return blend; + } + + private static AttachmentBlend ReplaceBlend(bool enabled) + { + AttachmentBlend blend = AttachmentBlend.Default; + blend.Enabled = enabled; + blend.ColorOp = BlendOp.Add; + blend.AlphaOp = BlendOp.Add; + blend.SrcColor = BlendFactor.One; + blend.DstColor = BlendFactor.Zero; + blend.SrcAlpha = BlendFactor.One; + blend.DstAlpha = BlendFactor.Zero; + return blend; + } + + private bool IsPrimaryTarget(FrameBufferRef target) => + FrameBuffers != null && FrameBuffers.Count > 0 && ReferenceEquals(target, FrameBuffers[0]); + + private bool IsTransparentTarget(FrameBufferRef target) => + FrameBuffers != null && FrameBuffers.Count > 1 && ReferenceEquals(target, FrameBuffers[1]); + + // -------------------------------------------------------------------- the textures + + /// + /// The textures this draw samples: every sampler the program declares, resolved to the + /// handle the client last pointed it at, with the program's own sampler override where it + /// has one (the terrain atlas read twice, nearest and linear, is exactly that case). + /// Returns how many entries of the scratch arrays are in use, and whether any of them is + /// the depth attachment of the target being drawn into - chunkliquid's fade against the + /// depth it draws with writes off. + /// + private int CollectChunkTextures(ShaderProgramBase program, string[] names, FrameBufferRef target, + out bool samplesBoundDepth) + { + samplesBoundDepth = false; + if (chunkScopeTextures.Length < names.Length) + { + chunkScopeTextures = new NativeTexture[names.Length]; + chunkScopeReads = new int[names.Length]; + } + + nativeProgramTextures.TryGetValue(program.ProgramId, out Dictionary? bound); + for (int i = 0; i < names.Length; i++) + { + int textureId = 0; + if (bound != null) bound.TryGetValue(names[i], out textureId); + + SamplerState? sampling = null; + if (program.customSamplers.TryGetValue(names[i], out int samplerId) && + device.TryNativeSamplerState(samplerId, out SamplerState custom)) + { + sampling = custom; + } + + if (textureId > 0 && textureId == target.DepthTextureId && !chunkScopeDepthWrite) + { + samplesBoundDepth = true; + } + + chunkScopeTextures[i] = new NativeTexture(NativeSamplerSlot.None, textureId, sampling); + chunkScopeReads[i] = textureId; + } + return names.Length; + } + + private string[] ChunkSamplerNames(int programId) + { + if (nativeProgramSamplers.TryGetValue(programId, out string[]? names)) return names; + names = device.SamplerNamesOf(programId); + nativeProgramSamplers[programId] = names; + return names; + } + + /// + /// A relinked or deleted program's cached sampler names, texture bindings and pipelines are + /// no longer its own: a shader reload hands the same id a different interface. + /// + internal void ForgetNativeChunkProgram(int programId) + { + nativeProgramSamplers.Remove(programId); + nativeProgramTextures.Remove(programId); + if (chunkPipelines.Count == 0) return; + + var stale = new List(); + foreach (KeyValuePair entry in chunkPipelines) + { + if (entry.Key.ProgramId == programId) stale.Add(entry.Key); + } + for (int i = 0; i < stale.Count; i++) chunkPipelines.Remove(stale[i]); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeEntities.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeEntities.cs new file mode 100644 index 00000000..b672c776 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeEntities.cs @@ -0,0 +1,348 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b decision 5 +// stage 2: the entity system on the native device API. +// +// What it draws: every entity's animated shape - the body through entityanimated in the Opaque +// stage (SystemRenderEntities.OnRenderOpaque3D's batched loop) and the same body through +// shadowmapentityanimated in the two shadow stages (SystemRenderEntities.OnRenderFrameShadows). +// Both reach here one sub-mesh at a time through ClientPlatformAbstract.RenderEntityMesh, the +// seam RenderAPIBase.RenderMultiTextureMesh draws through. +// Where the other side is: the seam's neutral body, which is the RenderMesh(MeshRef) call it +// replaced and which the OpenGL path still runs (ClientPlatformWindows.RenderMesh -> +// GL.DrawElements). NativeEntitiesEnabled false takes that route on the Vulkan device too, +// which is what the differential tests compare against. +// Target and slots: whatever the stage bound. Opaque -> Primary, with every bound colour slot +// in scope; shadow -> FrameBuffers[11]/[12], which have no colour attachment at all. +// State that is not obvious: +// - the motion attachment is a colour WRITE MASK here, never a draw-buffer toggle (decision 4). +// OptimumMotionWriteActive is the platform's own window state, so the pipeline masks the +// motion slot off when the window is shut and gives it replace-blend (ONE, ZERO, FUNC_ADD) +// when it is open - exactly what ClientPlatformWindows.ApplyOptimumMotionBlendState does. +// A slot past the motion attachment is masked off as well, so nothing Vulkan leaves +// undefined reaches an attachment GL would have kept (rule 9). +// - the SSAO G-buffer slots (2 and 3, present only when Primary carries them) take replace-blend +// too, which is the branch GlToggleBlend takes under RenderSSAO; +// - colour 0 and the glow slot 1 take the standard alpha blend GlToggleBlend(true) sets, because +// the caller turned blending on before the loop; +// - cull off, depth test on, depth write on, compare LESS: what SystemRenderEntities sets +// immediately before both loops (GlDisableCullFace / GlEnableDepthTest / GlToggleBlend(true)) +// and what ChunkRenderer left for the shadow stage, stated outright rather than read back +// out of the tracker (decision 3); +// - the draw is recorded INSIDE the stage's own declared pass, not a pass of its own: it names +// BoundPassName() with every slot, so RenderTargetManager.DeclarePass coalesces and +// EndNativePass(keepScope: true) leaves the scope open. One entity per rendering scope would +// otherwise cost an end/begin pair per entity per frame. +// - the bone matrices are NOT re-uploaded here. EntityShapeRenderer keeps calling +// UBO.Update("Animation", ...), which lands in the device's animation storage ring with its +// per-(frame, version) snapshot dedup; the native draw passes the real mesh id into +// BindProgramSets, which is what makes the ring's snapshot resolve for this draw. +// What it deliberately does NOT take native, and why: +// - held items through the `standard` program (EntityShapeRenderer.RenderItem). Their cull mode +// is decided per draw by renderInfo.CullFaces inside the VSEssentials fork, through +// GlDisableCullFace/GlEnableCullFace, and no seam carries it; a native pipeline would have to +// read it back off the tracker, which decision 3 forbids. It needs a seam in the fork, which +// is that fork's own change. +// - the `instanced` program: ShaderPrograms.Instanced has no vanilla call site in this tree +// (registration and manifest only), so there is nothing to port. The entity renderers' +// instanced draws go through RenderMeshInstanced with mod-registered shaders. +// What pins it: NativeEntityDrawTests (old route against native route, colour and the motion +// attachment) and Optimum.Tests/native-world-systems-coverage-tests.cs (the lib seam). +public partial class VulkanClientPlatform +{ + /// + /// False runs the seam's neutral body - the OpenGL body's RenderMesh - on the Vulkan device + /// instead of the native pass: the old route the differential tests compare against, in the + /// pattern of and . + /// + // On by default; OPTIMUM_VK_NATIVE_ENTITIES=0 sends every entity draw to the neutral body. + internal bool NativeEntitiesEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_ENTITIES") != "0"; + + /// The programs this file owns. Anything else takes the seam's neutral body. + private const string EntityAnimatedPass = "entityanimated"; + + private const string EntityShadowPass = "shadowmapentityanimated"; + + /// + /// The texture the client declared for a sampler, or 0 - which resolves to the placeholder. + /// The table itself is in + /// VulkanClientPlatform.NativeChunks.cs, filled by NoteNativeProgramTexture from + /// BindProgramTexture2D/Cube - the client saying "this program's sampler is this texture". + /// A program whose draws are all native never runs the emulated resolve that fills the push + /// block's slots from the texture units, so the native draw has to resolve every sampler the + /// program declares, not only the one the seam names. + /// + internal int DeclaredProgramTexture(int programId, string samplerName) => + nativeProgramTextures.TryGetValue(programId, out Dictionary? samplers) && + samplers.TryGetValue(samplerName, out int textureId) + ? textureId + : 0; + + /// + /// The entity pipeline last handed out, with the state it was built for. The batched loop + /// draws every entity with the same program, target and mesh shape, so this hits on all but + /// the first draw of a stage and no draw allocates a description or a blend array. + /// + private readonly record struct NativeEntityKey( + int ProgramId, int FramebufferId, int LayoutId, int ColorCount, int MotionIndex, bool MotionOpen); + + private NativeEntityKey nativeEntityKey; + private NativePipeline? nativeEntityPipeline; + private NativeTexture[] nativeEntityTextures = Array.Empty(); + private int[] nativeEntityReads = Array.Empty(); + private bool nativeEntityReported; + + /// An entity's shape: the native draw inside the stage's pass, or the neutral body. + public override void RenderEntityMesh(MeshRef mesh, string samplerName, int textureId) + { + if (!NativeEntitiesEnabled || device == null || mesh == null) + { + base.RenderEntityMesh(mesh!, samplerName, textureId); + return; + } + + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + FrameBufferRef target = CurrentFrameBuffer; + var vao = mesh as VAO; + if (program == null || target == null || vao == null || vao.VaoId == 0 || vao.Disposed || + !IsNativeEntityProgram(program)) + { + base.RenderEntityMesh(mesh, samplerName, textureId); + return; + } + + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) + { + base.RenderEntityMesh(mesh, samplerName, textureId); + return; + } + + NativePipeline? pipeline = NativeEntityPipelineFor(program, target, layoutId); + if (pipeline == null) + { + base.RenderEntityMesh(mesh, samplerName, textureId); + return; + } + + // Every sampler the program declares, resolved from what the client declared for it, with + // this draw's own texture for the sampler the seam names. + ResolveNativeEntityTextures(pipeline, program.ProgramId, samplerName, textureId); + + RuntimeStats.drawCallsCount++; + if (device.BeginNativePass(new NativePassDescription + { + // The stage's own pass, so the declaration coalesces and the scope stays open across + // the whole loop. Closed with keepScope below for the same reason. + Name = BoundPassName(), + FramebufferId = target.FboId, + ColorSlots = uint.MaxValue, + Reads = nativeEntityReads, + Flags = passContextFlags, + })) + { + device.DrawNativeMesh(pipeline, vao.VaoId, nativeEntityTextures); + } + device.EndNativePass(keepScope: true); + } + + /// Whether the program in use is one this file draws natively. + private static bool IsNativeEntityProgram(ShaderProgramBase program) + { + if (!string.Equals(program.PassName, EntityAnimatedPass, StringComparison.Ordinal) && + !string.Equals(program.PassName, EntityShadowPass, StringComparison.Ordinal)) + { + return false; + } + + // A sampler the client gave its own filtering or wrap mode to is bound through the unit's + // sampler override, which a native draw does not read. Neither vanilla entity program does + // that; if one ever did, the neutral body keeps it correct instead of silently losing it. + if (program.customSamplers.Count != 0 || program.clampTToEdge) return false; + + // The vanilla programs, and the program the shader registry holds under the same pass + // name: VSEssentials' first-person hands (ModSystemFpHands.fpModeHandShader) registers its + // own entityanimated (ALLOWDEPTHOFFSET, its own Animation and, under TAA, AnimationPrev + // blocks), which replaces the registry entry. On 2026-09-16 that program drew the arm + // several times too large through this route with TAA on, with every bound input traced + // equal; on 2026-09-17 the same repro (OPTIMUM_VK_NATIVE_ENTITIES=all, + // OPTIMUM_VK_NATIVE_SHADERS=force, TAA on, headless) no longer showed it - the parity dump + // of the hand region matched the neutral body (depth identical, motion within 0.0023) - + // after the native routes that ran around it had changed. The cause was never named. + // Anything else registered under these names stays on the neutral body (decision 1); + // OPTIMUM_VK_NATIVE_ENTITIES=all admits every program with the pass name. + if (!AllEntityPrograms && + !ReferenceEquals(program, ShaderPrograms.Entityanimated) && + !ReferenceEquals(program, ShaderPrograms.Shadowmapentityanimated) && + !IsRegistryProgram(program)) + { + return false; + } + return true; + } + + /// + /// Whether the shader registry holds this program under its pass name. Only a program the + /// registry registered (PassId set, from 1) is looked up: ShaderRegistry's type initializer + /// publishes uncompiled programs into ShaderPrograms.*, so a program the registry never saw + /// must not be the first thing to touch it - that is what crashed the OpenGL loading screen + /// when the upscaler stand-down called into it during startup. + /// + internal static bool IsRegistryProgram(ShaderProgramBase program) => + program.PassId > 0 && program.PassName != null && + ReferenceEquals(program, ShaderRegistry.getProgramByName(program.PassName)); + + private static readonly bool AllEntityPrograms = + Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_ENTITIES") == "all"; + + /// + /// The pipeline for this program, target, mesh shape and motion-window state, rebuilt only + /// when one of those changes. Every piece of fixed state is stated from client state: the + /// stage's own toggles, the platform's motion window, and the target's attachment count. + /// + private NativePipeline? NativeEntityPipelineFor(ShaderProgramBase program, FrameBufferRef target, int layoutId) + { + int colorCount = target.ColorTextureIds?.Length ?? 0; + int motionIndex = MotionAttachmentIndex; + if (motionIndex < 0 || motionIndex >= colorCount) motionIndex = -1; + bool motionOpen = motionIndex >= 0 && OptimumMotionWriteActive; + + var key = new NativeEntityKey(program.ProgramId, target.FboId, layoutId, colorCount, motionIndex, motionOpen); + if (nativeEntityPipeline != null && key.Equals(nativeEntityKey) && + device!.IsNativePipelineLive(nativeEntityPipeline)) + { + return nativeEntityPipeline; + } + + RenderTargetFormats? formats = device!.NativeTargetFormats(target.FboId, uint.MaxValue); + if (formats == null) return null; + + var description = new NativePipelineDescription + { + ProgramId = program.ProgramId, + PassName = program.PassName, + VertexLayoutId = layoutId, + Blend = NativeEntityBlend(formats.ColorFormats.Length, motionIndex, motionOpen), + // SystemRenderEntities sets these immediately before both loops; ClientMain set the + // depth function to LESS for the whole 3D render. + DepthTest = true, + DepthWrite = true, + DepthCompare = CompareOp.Less, + Cull = CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + Targets = formats, + }; + + NativePipeline? pipeline = device.RequestNativePipeline(description, out string error); + if (pipeline == null) + { + if (!nativeEntityReported) + { + nativeEntityReported = true; + Logger.Warning("Optimum: no native pipeline for '{0}': {1}", program.PassName, error); + } + nativeEntityPipeline = null; + return null; + } + + nativeEntityReported = false; + nativeEntityKey = key; + nativeEntityPipeline = pipeline; + return pipeline; + } + + /// + /// The per-attachment blend the OpenGL body would be drawing with: standard alpha blending on + /// the shaded slots, replace on the SSAO G-buffer slots, replace on the motion attachment + /// while its window is open and no write at all when it is shut or the slot is past it. + /// + private static AttachmentBlend[] NativeEntityBlend(int colorCount, int motionIndex, bool motionOpen) + { + var blend = new AttachmentBlend[Math.Max(colorCount, 1)]; + // The attachments the shading pass writes: everything before the motion attachment, or + // every bound slot when there is none. 2 without the SSAO G-buffer, 4 with it. + int shaded = motionIndex >= 0 ? motionIndex : colorCount; + for (int i = 0; i < blend.Length; i++) + { + AttachmentBlend attachment = AttachmentBlend.Default; + if (i == motionIndex) + { + if (!motionOpen) + { + attachment.WriteMask = 0; + } + else + { + Replace(ref attachment); + } + } + else if (i >= shaded) + { + // Past the motion attachment: nothing the entity programs declare an output for. + attachment.WriteMask = 0; + } + else if (i >= 2) + { + // The SSAO G-buffer's normal and position slots: GlToggleBlend's RenderSSAO branch. + Replace(ref attachment); + } + else + { + // Colour and glow: GlToggleBlend(true), EnumBlendMode.Standard. + attachment.Enabled = true; + attachment.SrcColor = BlendFactor.SrcAlpha; + attachment.DstColor = BlendFactor.OneMinusSrcAlpha; + attachment.ColorOp = BlendOp.Add; + attachment.SrcAlpha = BlendFactor.SrcAlpha; + attachment.DstAlpha = BlendFactor.OneMinusSrcAlpha; + attachment.AlphaOp = BlendOp.Add; + } + blend[i] = attachment; + } + return blend; + } + + /// (ONE, ZERO) with FUNC_ADD on every channel: the source wins, blending or not. + private static void Replace(ref AttachmentBlend attachment) + { + attachment.Enabled = true; + attachment.SrcColor = BlendFactor.One; + attachment.DstColor = BlendFactor.Zero; + attachment.ColorOp = BlendOp.Add; + attachment.SrcAlpha = BlendFactor.One; + attachment.DstAlpha = BlendFactor.Zero; + attachment.AlphaOp = BlendOp.Add; + } + + /// + /// This draw's sampled textures: the seam's texture for the sampler it names, and what the + /// client declared for every other sampler the program has. Reused buffers, because the + /// batched loop calls this once per entity. + /// + private void ResolveNativeEntityTextures(NativePipeline pipeline, int programId, string samplerName, int textureId) + { + string[] names = pipeline.SamplerNames; + if (nativeEntityTextures.Length != names.Length) + { + nativeEntityTextures = new NativeTexture[names.Length]; + nativeEntityReads = new int[names.Length]; + } + for (int i = 0; i < names.Length; i++) + { + int id = string.Equals(names[i], samplerName, StringComparison.Ordinal) + ? textureId + : DeclaredProgramTexture(programId, names[i]); + nativeEntityTextures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + nativeEntityReads[i] = id; + } + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeGui.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeGui.cs new file mode 100644 index 00000000..74d20ff1 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeGui.cs @@ -0,0 +1,353 @@ +using System; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b decision 5 +// stage 2: the GUI and text systems whose fixed state is stated at their call site. +// +// Two of the systems in that group draw through the native device API here. The rest of the +// group - Render2DTexture's gui quads, guigear, the block highlights, the wireframe cube and +// the camera path - take the generic stated route (NativeStated.cs), and the reason is written down in +// docs/vulkan.md: their blend and depth state is not the +// caller's, it is whatever the frame left on the tracker, and the same Render2DTexture call is +// reached both with standard alpha and with premultiplied alpha (RenderAPIGame's +// Render2DTexturePremultipliedAlpha brackets it with GlToggleBlend). Decision 3 forbids a +// native pass from reading that back off tracked GL state, so those systems move once their +// blend mode is stated at the seam, which is a change across the GUI element tree and its own +// piece of work. +// +// What these two draw: +// - RenderTextureQuad: the unit quad ClientMain.RenderTextureIntoFrameBuffer stretches one +// texture's rectangle over another's - how every Cairo-drawn GUI and text surface is baked +// into a texture. It is the highest-frequency GUI draw there is, and the one the stage +// brief flags for descriptor churn: a native draw resolves its texture into the device's +// per-frame bindless arena (VulkanDevice.BeginNativeDraw -> BindlessTextureTable.Resolve), +// so a fresh Cairo texture costs one slot in this frame's arena and no permanent +// descriptor, which is exactly what the emulated unit route could not promise. +// - RenderOverlayLines: SystemRenderPlayerAimAcc's aiming reticle, five line-topology draws +// through the gui program with noTexture set, at two line widths. +// Where the other side is: ClientPlatformAbstract.RenderTextureQuad and +// ClientPlatformAbstract.RenderOverlayLines, whose neutral bodies are the RenderMesh calls +// these seams replaced and which the OpenGL path still runs (ClientPlatformWindows.RenderMesh +// -> GL.DrawElements). NativeGuiEnabled false takes that route on the Vulkan device too, which +// is what the differential tests compare against. +// Target and slots: CurrentFrameBuffer, which for the texture blit is the framebuffer the +// caller just bound over the destination texture and for the reticle is the default +// framebuffer the Ortho stage draws into. Every bound colour slot is in the pass, so the scope +// is the one the emulated draw opens; both fragment shaders write outColor at 0 only and the +// pipeline masks every other slot off (rule 9). Neither writes depth, and neither writes the +// motion attachment - the Ortho stage runs after the TAA resolve, so there is no motion slot +// to mask in the first place. +// State that is not obvious: +// - no depth test and no depth write: RenderTextureIntoFrameBuffer calls GlDisableDepthTest +// itself, and the reticle is a 2D overlay whose ortho projection puts it in front; +// - blend: the value the caller computed, through the same factor table the tracker uses +// (AttachmentBlend.For), never read back off the tracker; +// - no culling: both meshes are screen-facing quads and lines, which GL rasterizes whatever +// the cull state, and lines are not culled at all; +// - the topology is the mesh's own draw mode (VulkanDevice.NativeMeshTopology), because that +// is where the tesselator put it - EnumDrawMode.Lines for the reticle - not a state toggle; +// - the line width is the caller's, and it is the one piece of dynamic state a fullscreen +// pass never needed; it is in the native pipeline key, so the 0.5 and the 1.0 draws of one +// frame do not share a pipeline. +// What pins it: NativeGuiTests (old route against native route, both systems, both line +// widths, blend on and off) and Optimum.Tests/native-world-systems-coverage-tests.cs (the lib +// seams). +public partial class VulkanClientPlatform +{ + /// + /// False runs the seams' neutral bodies - the OpenGL bodies' RenderMesh - on the Vulkan + /// device instead of the native passes: the old route the differential tests compare + /// against, in the pattern of . + /// + internal bool NativeGuiEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_GUI") != "0"; + + /// + /// The texture-into-texture blit's pipeline and placements. Nothing is written per draw: + /// the source rectangle, the destination rectangle and the alpha test are all in the + /// program's own shadow by the time the seam is reached, put there by the setters + /// RenderTextureIntoFrameBuffer calls. + /// + private readonly NativeMeshPass nativeTextureQuad = + new("texture2texture", Array.Empty(), new[] { "tex2d" }); + + /// + /// The 2D line overlay's pipeline and placements, on the gui program. "tex2dOverlay" is + /// resolved as well as "tex2d" so neither sampler slot keeps a stale bindless index out of + /// the push shadow; the reticle sets noTexture, so gui.fsh samples neither. + /// + private readonly NativeMeshPass nativeOverlayLines = + new("gui", Array.Empty(), new[] { "tex2d", "tex2dOverlay" }); + + /// + /// The GUI quads' pipeline and placements, on the gui program - a pass of its own so the + /// reticle's line pipeline and these triangle pipelines do not evict each other. + /// + private readonly NativeMeshPass nativeGuiQuad = + new("gui", Array.Empty(), new[] { "tex2d", "tex2dOverlay" }); + + /// + /// One Render2DTexture quad: the native pass, or the neutral body's RenderMesh. + /// What it draws: a GUI element's texture. The other side: ClientPlatformAbstract.RenderGuiQuad. + /// Target and slots: the caller's target, slot 0. State: the blend mode, depth test, depth + /// mask, depth function and scissor the caller last stated through this platform's virtuals + /// (VulkanClientPlatform.State.cs), so premultiplied-alpha blits and scrolled, clipped lists + /// draw as they do on GL. Only the vanilla gui program is taken. + /// + public override void RenderGuiQuad(MeshRef quad, int textureId) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeGuiEnabled || device == null || quad == null || program == null || + !ReferenceEquals(program, ShaderPrograms.Gui) || + !DrawNativeGuiMesh(nativeGuiQuad, quad, textureId, + DeclaredProgramTexture(program.ProgramId, "tex2dOverlay"), 1.0f, + statedBlendOn, statedBlendMode, statedDepthTest, statedDepthWrite, + GlEnums.CompareOpFrom(statedDepthFunc), scissorEnabled ? statedScissor : null, "GuiQuad")) + { + base.RenderGuiQuad(quad!, textureId); + } + } + + /// + /// Any other draw under the vanilla gui program - block highlights, the wireframe, gear and + /// progress overlays, mods drawing with the gui shader through IRenderAPI.RenderMesh - through + /// its own pass cache, so line and triangle pipelines of different callers do not evict the + /// quads'. + /// + private readonly NativeMeshPass nativeParticles2d = + new("particlesquad2d", Array.Empty(), new[] { "particleTex" }); + + /// + /// The main menu's 2D particle pool (ParticleRenderer2D.Render -> RenderMeshInstanced under the + /// vanilla particlesquad2d program) as a native instanced pass on whatever target is bound - + /// the default framebuffer in the menu. The OpenGL side is ClientPlatformWindows.RenderMeshInstanced. + /// State is what the client stated: blend on in the non-OIT mode (GlToggleBlend), depth test and + /// mask as left by the menu. particleTex resolves from the program's declared texture. + /// + private bool TryRenderParticles2dNative(MeshRef mesh, int quantity) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeGuiEnabled || device == null || mesh == null || program == null || quantity <= 0 || + !ReferenceEquals(program, ShaderPrograms.Particlesquad2d)) + { + return false; + } + + return DrawNativeGuiMesh(nativeParticles2d, mesh, + DeclaredProgramTexture(program.ProgramId, "particleTex"), 0, + statedLineWidth, statedBlendOn, statedBlendMode, statedDepthTest, statedDepthWrite, + GlEnums.CompareOpFrom(statedDepthFunc), scissorEnabled ? statedScissor : null, "Particles2d", + CullModeFlags.None, quantity); + } + + private readonly NativeMeshPass nativeMinimalGui = + new("", Array.Empty(), new[] { "tex2d" }); + + private readonly NativeMeshPass nativeGuiGear = + new("guigear", Array.Empty(), new[] { "tex2d" }); + + /// + /// Single-sampler GUI quads drawn through plain RenderMesh. The temporal stability gear: + /// HudHotbar draws capi.Gui.QuadMeshRef under the vanilla guigear program in the Ortho stage. + /// The early loading screen's quads: MainMenuRenderAPI.Render2DTexture draws through the + /// platform's hardcoded ShaderProgramMinimalGui (no pass name, no asset) until the shader + /// registry is up, and that RenderMesh lands here. The OpenGL side is + /// ClientPlatformWindows.RenderMesh. One sampler (tex2d), the state the client stated. + /// + private bool TryRenderMinimalGuiNative(MeshRef mesh) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeGuiEnabled || device == null || mesh == null || program == null) return false; + + NativeMeshPass pass; + string label; + if (ReferenceEquals(program, MinimalGuiShader)) + { + pass = nativeMinimalGui; + label = "MinimalGui"; + } + else if (ReferenceEquals(program, ShaderPrograms.Guigear)) + { + pass = nativeGuiGear; + label = "GuiGear"; + } + else + { + return false; + } + + return DrawNativeGuiMesh(pass, mesh, + DeclaredProgramTexture(program.ProgramId, "tex2d"), 0, + statedLineWidth, statedBlendOn, statedBlendMode, statedDepthTest, statedDepthWrite, + GlEnums.CompareOpFrom(statedDepthFunc), scissorEnabled ? statedScissor : null, label); + } + + private readonly NativeMeshPass nativeGuiMesh = + new("gui", Array.Empty(), new[] { "tex2d", "tex2dOverlay" }); + + /// + /// A plain RenderMesh under the vanilla gui program, recorded natively under the state the + /// client stated: blend, depth, depth function, scissor, cull, line width. The sampled + /// textures are the program's declared ones. False: the caller runs the generic stated draw. + /// Called from VulkanClientPlatform.RenderMesh; the seams' neutral bodies reach it too, which + /// is why it honours itself. + /// + private bool TryRenderGuiMeshNative(MeshRef mesh) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeGuiEnabled || device == null || mesh == null || program == null || + !ReferenceEquals(program, ShaderPrograms.Gui)) + { + return false; + } + CullModeFlags cull = statedCull + ? (statedCullBack ? CullModeFlags.BackBit : CullModeFlags.FrontBit) + : CullModeFlags.None; + return DrawNativeGuiMesh(nativeGuiMesh, mesh, + DeclaredProgramTexture(program.ProgramId, "tex2d"), + DeclaredProgramTexture(program.ProgramId, "tex2dOverlay"), + statedLineWidth, statedBlendOn, statedBlendMode, statedDepthTest, statedDepthWrite, + GlEnums.CompareOpFrom(statedDepthFunc), scissorEnabled ? statedScissor : null, "GuiMesh", cull); + } + + /// The texture-into-texture blit's draw: the native pass, or the neutral body's RenderMesh. + public override void RenderTextureQuad(MeshRef quad, int textureId, bool blend) + { + if (!NativeGuiEnabled || device == null || quad == null) + { + base.RenderTextureQuad(quad!, textureId, blend); + return; + } + + if (!DrawNativeGuiMesh(nativeTextureQuad, quad, textureId, 0, 1.0f, blend, "TextureQuad")) + { + base.RenderTextureQuad(quad, textureId, blend); + } + } + + /// The 2D line overlay's draw: the native pass, or the neutral body's RenderMesh. + public override void RenderOverlayLines(MeshRef lines, int textureId, float lineWidth, bool blend) + { + if (!NativeGuiEnabled || device == null || lines == null) + { + base.RenderOverlayLines(lines!, textureId, lineWidth, blend); + return; + } + + if (!DrawNativeGuiMesh(nativeOverlayLines, lines, textureId, 0, lineWidth, blend, "OverlayLines")) + { + base.RenderOverlayLines(lines, textureId, lineWidth, blend); + } + } + + /// + /// One GUI mesh recorded as its own native pass: the target the caller bound, every bound + /// colour slot in scope, the fixed state the caller stated, the mesh's own vertex layout + /// and topology, and the sampled textures resolved from handles. + /// + /// False means nothing was recorded and the caller must run the seam's neutral body - a + /// missing target, a program that is not the one this pass is for, a mesh the device does + /// not know, or a pipeline that is still compiling. + /// + private bool DrawNativeGuiMesh(NativeMeshPass pass, MeshRef mesh, int textureId, int overlayTextureId, + float lineWidth, bool blend, string passLabel) + => DrawNativeGuiMesh(pass, mesh, textureId, overlayTextureId, lineWidth, blend, EnumBlendMode.Standard, + depthTest: false, depthWrite: false, CompareOp.Less, scissor: null, passLabel, CullModeFlags.None); + + private bool DrawNativeGuiMesh(NativeMeshPass pass, MeshRef mesh, int textureId, int overlayTextureId, + float lineWidth, bool blend, EnumBlendMode blendMode, bool depthTest, bool depthWrite, CompareOp depthCompare, + Rect2D? scissor, string passLabel, CullModeFlags cull = CullModeFlags.None, int instanceCount = 1) + { + FrameBufferRef target = CurrentFrameBuffer; + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + var vao = mesh as VAO; + if (program == null || vao == null || vao.VaoId == 0 || vao.Disposed) return false; + // ShaderProgramMinimalGui has no pass name; its pass is named "". + if (!string.Equals(program.PassName ?? "", pass.PassName, StringComparison.Ordinal)) return false; + + // The Ortho stage draws into the default framebuffer, which has no FrameBufferRef of + // its own (ClientPlatformWindows.LoadFrameBuffer sets CurrentFrameBuffer null for it); + // the pass API names it the same way the post chain does. + int framebufferId = target?.FboId ?? PassDeclaration.DefaultFramebuffer; + uint slots = target != null ? NativeAllColorSlots(target) : 1u; + + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) return false; + + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, slots); + if (formats == null) return false; + + NativePipeline? pipeline = NativeMeshPipelineFor(pass, program, framebufferId, slots, layoutId, + new NativePipelineDescription + { + Blend = GuiSlots(formats, blend, framebufferId, blendMode), + DepthTest = depthTest, + DepthWrite = depthWrite, + DepthCompare = depthCompare, + Cull = cull, + Topology = device.NativeMeshTopology(vao.VaoId), + LineWidth = lineWidth, + }); + if (pipeline == null) return false; + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + Rect2D viewport = StatedViewport(); + bool recorded = false; + if (device.BeginNativePass(new NativePassDescription + { + Name = passLabel + "/" + framebufferId, + FramebufferId = framebufferId, + ColorSlots = slots, + Reads = pass.SamplerNames.Length > 1 + ? new[] { textureId, overlayTextureId } + : new[] { textureId }, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + Scissor = scissor, + })) + { + Span textures = stackalloc NativeTexture[pass.Samplers.Length]; + textures[0] = new NativeTexture(pass.Samplers[0], textureId); + if (textures.Length > 1) textures[1] = new NativeTexture(pass.Samplers[1], overlayTextureId); + recorded = device.DrawNativeMeshInstanced(pipeline, vao.VaoId, instanceCount, textures); + } + device.EndNativePass(); + + // Whatever the stage was drawing into before this pass keeps drawing into it through + // the generic stated route, so its own pass context is restored - the same restore the + // sky pass and the TAA resolve do. + SetPassContext(outer, outerFlags); + return recorded; + } + + /// + /// A 2D overlay's fixed state per colour attachment: the caller's blend on slot 0 through + /// the tracker's own factor table, and every other slot masked off so an attachment the + /// fragment shader never writes keeps its contents as it does on GL (rule 9). + /// + private AttachmentBlend[] GuiSlots(RenderTargetFormats formats, bool blend, int framebufferId, + EnumBlendMode mode = EnumBlendMode.Standard) + { + var slots = new AttachmentBlend[Math.Max(formats.ColorFormats.Length, 1)]; + slots[0] = AttachmentBlend.For(blend, mode); + // World/UI separation: straight-alpha GUI accumulates coverage in the UI image. + if (stated.IsUiImage(framebufferId)) slots[0] = slots[0].ForUiImage(); + for (int i = 1; i < slots.Length; i++) + { + slots[i] = AttachmentBlend.Default; + slots[i].WriteMask = 0; + } + return slots; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostChain.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostChain.cs new file mode 100644 index 00000000..a7769330 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostChain.cs @@ -0,0 +1,683 @@ +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), stage 1: Optimum owns +// the post and TAA chain end to end. This file holds the chain - its ORDER, and one helper per +// pass with a stable signature. +// +// The order is section 3's and the OpenGL body's: OIT merge, sky motion, SSAO and blur then the +// AO composite, TAA resolve, bloom, god rays, FXAA luma or blit, final composition, late overlays, +// TAA sharpen at the blit boundary, and last the blit/FSR/debug step that is already native +// (VulkanClientPlatform.NativeBlit.cs). +// RenderPostprocessingEffects' override runs the steps that live inside it and never calls base. +// +// Four helpers draw natively here - the OIT merge, sky motion and the two TAA passes - through RequestNativePipeline, +// BeginNativePass, WriteNative and DrawNativeFullscreen, exactly as the blit does. Every other +// helper is LEGACY: the same work through the GL-shaped platform calls, which after the split in +// ClientPlatformWindows is one lib virtual per pass, so the chain is complete and correct at +// every commit and a later stage replaces one helper at a time. The bloom chain, god rays, the +// Luma step and the final composition are native in VulkanClientPlatform.NativePostFinal.cs; +// their LEGACY helpers stay as the old route the differential tests compare against. +public partial class VulkanClientPlatform +{ + /// The chain's passes, in the order the frame runs them (section 3). + internal enum NativePostStep + { + OitMerge = 0, + SkyMotion = 1, + SsaoAndAmbientOcclusion = 2, + TaaResolve = 3, + TaaSharpen = 4, + Bloom = 5, + GodRays = 6, + FxaaOrBlit = 7, + FinalComposition = 8, + Blit = 9, + } + + /// + /// The chain order, declared once. The steps the chain itself sequences are the ones inside + /// ; the others are separate virtuals the client + /// calls in this order, and every helper records its step, so a test can read the order back + /// off a real frame. + /// + internal static readonly NativePostStep[] NativePostChainOrder = + { + NativePostStep.OitMerge, + NativePostStep.SkyMotion, + NativePostStep.SsaoAndAmbientOcclusion, + NativePostStep.TaaResolve, + NativePostStep.Bloom, + NativePostStep.GodRays, + NativePostStep.FxaaOrBlit, + NativePostStep.FinalComposition, + NativePostStep.TaaSharpen, + NativePostStep.Blit, + }; + + /// + /// False runs the whole chain on the OpenGL body instead - every pass, the blit included: + /// the old route the differential tests compare the native one against. The blit keeps its + /// own switch for the tests written around it. + /// + internal bool NativePostChainEnabled { get; set; } = true; + + /// Test seam: when set, every chain step appends itself here as it runs. + internal List? NativePostStepLog { get; set; } + + private bool UseNativePostChain => NativePostChainEnabled && device != null; + + private void NotePostStep(NativePostStep step) => NativePostStepLog?.Add(step); + + /// + /// Test seam: one chain step on whichever route selects, + /// without the eight steps around it. The differential tests run the same inputs through both. + /// + internal void RunPostStepAmbientOcclusionForTests(float[] projectMatrix) => + PostStepAmbientOcclusion(projectMatrix); + + // ------------------------------------------------------------------ the chain + + /// + /// The chain's own steps, in order, with no call to the base body: the AO step, the TAA + /// resolve, bloom, god rays, the Luma step and the epilogue. TAA sharpen runs after the + /// separate final-composition call, so those effects consume the unsharpened resolve. The order and the + /// per-step conditions are the OpenGL body's + /// (ClientPlatformWindows.RenderPostprocessingEffects). + /// + private void RunNativePostChain(float[] projectMatrix) + { + if (!offscreenBufferActive) return; + + SetPassContext("Post", PassFlags.None); + NotePostStep(NativePostStep.SsaoAndAmbientOcclusion); + PostStepAmbientOcclusion(projectMatrix); + + NotePostStep(NativePostStep.TaaResolve); + PostStepTaaResolve(); + + int scene = OptimumPostSceneTexture(); + int glow = OptimumPostGlowTexture(); + + NotePostStep(NativePostStep.Bloom); + PostStepBloom(scene, glow); + NotePostStep(NativePostStep.GodRays); + PostStepGodRays(scene, glow); + NotePostStep(NativePostStep.FxaaOrBlit); + PostStepFxaaOrBlit(scene); + PostStepFinish(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + // ----------------------------------------------------------- native pass 1: OIT merge + + private readonly NativeFullscreenPass nativeOitMerge = new("transparentcompose", + Array.Empty(), + new[] { "accumulation", "revealage", "inGlow", "OITreveal", "OITaccumulation" }); + + /// + /// The OIT merge, drawn natively. The OpenGL body binds Primary without touching the + /// viewport, turns the depth test off and blending on with the global source-alpha mode, + /// opens the motion window when TAA is running, and composes the Transparent target's three + /// attachments together with the OIT reveal and accumulation targets. + /// + /// Here the pass states its target, its colour slots and its reads, and the blend is the + /// pipeline's: source-alpha on every slot, and additive (ONE, ONE) on the motion attachment + /// alone, which is what lets the merge add the transparent layer's coverage into the + /// reactive channel without touching the vector or the writer depth under it. The GL-shaped + /// state calls stay, outside the pass: they are what every stage AFTER the merge inherits, + /// exactly as on the OpenGL body. + /// + private void NativeOitMerge() + { + List buffers = FrameBuffers; + FrameBufferRef primary = buffers != null && buffers.Count > 0 ? buffers[0] : null!; + FrameBufferRef transparent = buffers != null && buffers.Count > 1 ? buffers[1] : null!; + ShaderProgramTransparentcompose compose = ShaderPrograms.Transparentcompose; + if (!offscreenBufferActive || primary == null || transparent == null || + transparent.ColorTextureIds == null || transparent.ColorTextureIds.Length < 3 || + compose == null || compose.LoadError || compose.Disposed) + { + LegacyOitMerge(); + return; + } + + OptimumBindKeepViewport(primary); + ApplyTransparentMergeBlendState(); + + bool motion = NativeMotionAttachmentWritable(primary); + uint slots = NativeWorldColorSlots(); + if (motion) slots |= 1u << MotionAttachmentIndex; + + NativePipeline? pipeline = NativePostPipeline(nativeOitMerge, compose, primary.FboId, slots, + NativeMergeBlend(slots, motion), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) + { + LegacyOitMerge(); + return; + } + + int accumulation = transparent.ColorTextureIds[0]; + int revealage = transparent.ColorTextureIds[1]; + int inGlow = transparent.ColorTextureIds[2]; + int oitReveal = SystemRenderOITLayers.OptimumOitRevealTexture; + int oitAccumulation = SystemRenderOITLayers.OptimumOitAccumTexture; + + var reads = new List { accumulation, revealage, inGlow }; + if (oitReveal > 0) reads.Add(oitReveal); + if (oitAccumulation > 0) reads.Add(oitAccumulation); + + if (BeginNativeKeepViewportPass("MergeTransparent/0", primary.FboId, slots, reads.ToArray())) + { + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeOitMerge.Samplers[0], accumulation), + new NativeTexture(nativeOitMerge.Samplers[1], revealage), + new NativeTexture(nativeOitMerge.Samplers[2], inGlow), + new NativeTexture(nativeOitMerge.Samplers[3], oitReveal), + new NativeTexture(nativeOitMerge.Samplers[4], oitAccumulation), + }); + } + device.EndNativePass(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + /// + /// The merge's blend: the global source-alpha mode on every slot the pass writes, with + /// FUNC_ADD and (ONE, ONE) on the motion attachment while the window is open + /// (ClientPlatformWindows.ApplyOptimumMotionAccumulateBlendState). + /// + private AttachmentBlend[] NativeMergeBlend(uint slots, bool motion) + { + var blend = new AttachmentBlend[NativeSlotCount(slots)]; + for (int i = 0; i < blend.Length; i++) + { + if (((slots >> i) & 1) == 0) + { + blend[i].WriteMask = 0; + continue; + } + AttachmentBlend attachment = AttachmentBlend.Default; + attachment.Enabled = true; + if (motion && i == MotionAttachmentIndex) + { + attachment.SrcColor = BlendFactor.One; + attachment.DstColor = BlendFactor.One; + attachment.SrcAlpha = BlendFactor.One; + attachment.DstAlpha = BlendFactor.One; + } + blend[i] = attachment; + } + return blend; + } + + // ---------------------------------------------------------- native pass 2: sky motion + + private readonly NativeFullscreenPass nativeSkyMotion = new("taa-skymotion", + new[] { "taaRenderSize", "taaJitterPx", "taaInvViewProjJittered", "taaPrevViewProj", "taaCloudReactive" }, + new[] { "transparentRevealTex" }); + + /// + /// The sky / volumetric-cloud motion and reactive pass, drawn natively: a fullscreen + /// triangle at window depth 1.0 with the depth test on, GL_LEQUAL and depth writes off, + /// writing the motion attachment and nothing else. The guards, the two matrices and the + /// uniform values are the OpenGL body's (ClientPlatformWindows.RenderOptimumSkyMotion); the + /// motion-only window is the pass's colour slot, not a draw-buffer mask. + /// + private bool NativeSkyMotion() + { + if (!OptimumConfig.EffectiveTaa) return false; + if (!TaaTargetsReady || MotionAttachmentIndex < 0) return false; + ShaderProgram skyMotion = ShaderPrograms.TaaSkyMotion; + if (skyMotion == null || skyMotion.LoadError || skyMotion.Disposed) return false; + List buffers = FrameBuffers; + if (buffers == null || buffers.Count <= 1) return false; + FrameBufferRef primary = buffers[0]; + FrameBufferRef transparent = buffers[1]; + if (primary == null || primary.Disposed || transparent == null || transparent.Disposed) return false; + if (transparent.ColorTextureIds == null || transparent.ColorTextureIds.Length < 2) return false; + + OptimumTemporalFrame frame = OptimumTemporal.Frame; + if (!frame.WasViewCaptured(EnumTemporalView.World)) return false; + + // The same two matrices the resolve builds for its camera fallback: this frame's + // jittered view-projection inverted, and the previous frame's unjittered one. Built + // here rather than reused, exactly as the OpenGL body builds them. + float[] projection = frame.GetProjection(EnumTemporalView.World); + var jittered = new double[16]; + for (int i = 0; i < 16; i++) jittered[i] = projection[i]; + OptimumTemporalMath.ApplyProjectionJitter(jittered, frame.JitterPx.X, frame.JitterPx.Y, + primary.Width, primary.Height); + var projectionJittered = new float[16]; + for (int i = 0; i < 16; i++) projectionJittered[i] = (float)jittered[i]; + float[] viewProj = Mat4f.Mul(new float[16], projectionJittered, frame.CameraMatrixOrigin); + float[] invViewProj = Mat4f.Invert(new float[16], viewProj); + // A failed invert bails before any state is touched, as it does on the OpenGL body. + if (invViewProj == null) return false; + float[] prevViewProj = Mat4f.Mul(new float[16], + frame.GetPrevProjection(EnumTemporalView.World), frame.PrevCameraMatrixOrigin); + + bool opened = NativeMotionAttachmentWritable(primary); + if (opened) + { + uint slots = 1u << MotionAttachmentIndex; + int reveal = transparent.ColorTextureIds[1]; + NativePipeline? pipeline = NativePostPipeline(nativeSkyMotion, skyMotion, primary.FboId, slots, + NativeMotionOnlyBlend(slots), depthTest: true, depthWrite: false, CompareOp.LessOrEqual); + if (pipeline == null) + { + opened = false; + } + else + { + SetPassContext("SkyMotion", PassFlags.None); + if (BeginNativeKeepViewportPass("SkyMotion/0", primary.FboId, slots, new[] { reveal })) + { + device.WriteNative(pipeline, nativeSkyMotion.Uniforms[0], primary.Width, primary.Height); + device.WriteNative(pipeline, nativeSkyMotion.Uniforms[1], frame.JitterPx.X, frame.JitterPx.Y); + WriteNativeMatrix(pipeline, nativeSkyMotion.Uniforms[2], invViewProj); + WriteNativeMatrix(pipeline, nativeSkyMotion.Uniforms[3], prevViewProj); + device.WriteNative(pipeline, nativeSkyMotion.Uniforms[4], OptimumCloudReactive); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeSkyMotion.Samplers[0], reveal), + }); + } + device.EndNativePass(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + } + + // Everything the OpenGL body's state block and its finally leave for the AfterOIT + // stages and the post chain: depth writes on, GL_LESS, blending on, culling on. The + // body normalises them whether or not the window opened, so this runs on both paths. + GlDepthFunc(EnumDepthFunction.Less); + GlDepthMask(flag: true); + GlToggleBlend(on: true); + GlEnableCullFace(); + + if (!opened) return false; + ScreenManager.FrameProfiler.Mark("rend3D-ret-skymv"); + return true; + } + + /// The motion attachment alone, unblended; every other slot masked out of the pass. + private AttachmentBlend[] NativeMotionOnlyBlend(uint slots) => NativeOpaqueBlend(slots); + + /// + /// Unblended writes on every slot the pass owns, every other slot masked out of it: what a + /// fullscreen pass that owns its pixels asks for, and what the OpenGL body expresses as + /// blending off plus a draw-buffer mask. + /// + private AttachmentBlend[] NativeOpaqueBlend(uint slots) + { + var blend = new AttachmentBlend[NativeSlotCount(slots)]; + for (int i = 0; i < blend.Length; i++) + { + if (((slots >> i) & 1) == 0) blend[i].WriteMask = 0; + else blend[i] = AttachmentBlend.Default; + } + return blend; + } + + // ------------------------------------------------------- native passes 4 and 5: TAA + + private readonly NativeFullscreenPass nativeTaaResolve = new("taa-resolve", + new[] + { + "renderSize", "jitterPx", "invViewProjJittered", "prevViewProj", "viewMatrix", + "cameraDelta", "resetHistory", "blendAlpha", "varianceGamma", + }, + new[] { "sceneTex", "glowTex", "motionTex", "depthTex", "historyColor", "historyGlow", "historyDepth" }); + + private readonly NativeFullscreenPass nativeTaaSharpen = new("taa-sharpen", + new[] { "inputTexelSize", "sharpness" }, + new[] { "inputScene" }); + + /// + /// The TAA resolve's draw, drawn natively. Every decision stays in the lib body + /// (ClientPlatformWindows.RenderOptimumTaaResolve) - the guards, the jittered and previous + /// view-projections, the reset test, and afterwards the resolved textures, the history + /// validity and the parity flip - so the temporal contract is bit-identical whichever route + /// draws: this helper only replaces the draw. + /// + /// The history slot owns all three of its attachments (colour, glow and linear depth), so + /// the pass writes all three colour slots with no blending, no depth test and no depth + /// attachment, at the slot's own size. The seven inputs are the pass's declared reads and + /// resolve straight to bindless slots; the nine uniform values are the OpenGL body's, at + /// their placements. The resolve's temporal invariants live in taa-resolve.fsh, which both routes run + /// unchanged - the 3x3 nearest-depth disocclusion and the luminance anti-flicker weighting + /// are the shader's, and nothing here touches them. + /// + private void NativeTaaResolve(FrameBufferRef write, FrameBufferRef read, + float[] invViewProjJittered, float[] prevViewProj, bool reset) + { + List buffers = FrameBuffers; + FrameBufferRef primary = buffers != null && buffers.Count > 0 ? buffers[0] : null!; + ShaderProgram resolve = ShaderPrograms.TaaResolve; + if (primary == null || primary.Disposed || primary.ColorTextureIds == null || + MotionAttachmentIndex < 0 || primary.ColorTextureIds.Length <= MotionAttachmentIndex || + write.ColorTextureIds == null || write.ColorTextureIds.Length < 3 || + read.ColorTextureIds == null || read.ColorTextureIds.Length < 3 || + resolve == null || resolve.LoadError || resolve.Disposed) + { + base.OptimumTaaResolveDraw(write, read, invViewProjJittered, prevViewProj, reset); + return; + } + + // Colour, glow and linear depth: the three attachments the history framebuffer was + // built with, all written by this one draw. + const uint slots = 0b111u; + NativePipeline? pipeline = NativePostPipeline(nativeTaaResolve, resolve, write.FboId, slots, + NativeOpaqueBlend(slots), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) + { + base.OptimumTaaResolveDraw(write, read, invViewProjJittered, prevViewProj, reset); + return; + } + + int scene = primary.ColorTextureIds[0]; + int glow = primary.ColorTextureIds[1]; + int motion = primary.ColorTextureIds[MotionAttachmentIndex]; + int depth = primary.DepthTextureId; + int historyColor = read.ColorTextureIds[0]; + int historyGlow = read.ColorTextureIds[1]; + int historyDepth = read.ColorTextureIds[2]; + + OptimumTemporalFrame frame = OptimumTemporal.Frame; + if (BeginNativeTargetPass("TaaResolve/" + write.FboId, write.FboId, slots, + write.Width, write.Height, + new[] { scene, glow, motion, depth, historyColor, historyGlow, historyDepth })) + { + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[0], write.Width, write.Height); + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[1], frame.JitterPx.X, frame.JitterPx.Y); + WriteNativeMatrix(pipeline, nativeTaaResolve.Uniforms[2], invViewProjJittered); + WriteNativeMatrix(pipeline, nativeTaaResolve.Uniforms[3], prevViewProj); + WriteNativeMatrix(pipeline, nativeTaaResolve.Uniforms[4], frame.CameraMatrixOrigin); + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[5], + frame.CameraPosDelta.X, frame.CameraPosDelta.Y, frame.CameraPosDelta.Z); + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[6], reset ? 1 : 0); + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[7], 0.1f); + device.WriteNative(pipeline, nativeTaaResolve.Uniforms[8], 1.25f); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeTaaResolve.Samplers[0], scene), + new NativeTexture(nativeTaaResolve.Samplers[1], glow), + new NativeTexture(nativeTaaResolve.Samplers[2], motion), + new NativeTexture(nativeTaaResolve.Samplers[3], depth), + new NativeTexture(nativeTaaResolve.Samplers[4], historyColor), + new NativeTexture(nativeTaaResolve.Samplers[5], historyGlow), + new NativeTexture(nativeTaaResolve.Samplers[6], historyDepth), + }); + } + device.EndNativePass(); + FinishNativeTaaPass(); + } + + /// + /// The TAA sharpen's draw, drawn natively: one colour slot at the sharpen target's own + /// size, unblended, no depth. The conditions, the input texture (the final-composited Primary + /// colour) and the texture the pass hands on stay in the lib body + /// (ClientPlatformWindows.RenderOptimumTaaSharpen). + /// + private void NativeTaaSharpen(FrameBufferRef target, int resolvedScene) + { + ShaderProgram sharpen = ShaderPrograms.TaaSharpen; + if (target.ColorTextureIds == null || target.ColorTextureIds.Length < 1 || target.Disposed || + sharpen == null || sharpen.LoadError || sharpen.Disposed) + { + base.OptimumTaaSharpenDraw(target, resolvedScene); + return; + } + + NativePipeline? pipeline = NativePostPipeline(nativeTaaSharpen, sharpen, target.FboId, 1u, + NativeOpaqueBlend(1u), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) + { + base.OptimumTaaSharpenDraw(target, resolvedScene); + return; + } + + if (BeginNativeTargetPass("TaaSharpen/" + target.FboId, target.FboId, 1u, + target.Width, target.Height, new[] { resolvedScene })) + { + device.WriteNative(pipeline, nativeTaaSharpen.Uniforms[0], 1f / target.Width, 1f / target.Height); + device.WriteNative(pipeline, nativeTaaSharpen.Uniforms[1], + GameMath.Clamp(OptimumConfig.TaaSharpness, 0f, 1f)); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeTaaSharpen.Samplers[0], resolvedScene), + }); + } + device.EndNativePass(); + FinishNativeTaaPass(); + } + + /// + /// What both TAA draws hand back to the rest of the post chain, which is what the OpenGL + /// body's restore leaves: blending on in the standard mode, the depth test on and Primary + /// bound with its full-resolution viewport. Outside the native pass, on the GL-shaped + /// state, because that is what the steps after it read. + /// + private void FinishNativeTaaPass() + { + GlToggleBlend(on: true); + GlEnableDepthTest(); + LoadFrameBuffer(EnumFrameBuffer.Primary); + } + + // ------------------------------------------------------------------ legacy helpers + + /// LEGACY - pass 1 on the OpenGL body. replaces it; kept as the old route. + private void LegacyOitMerge() + { + SetPassContext("MergeTransparent", PassFlags.None); + base.MergeTransparentRenderPass(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + /// LEGACY - pass 2 on the OpenGL body. replaces it; kept as the old route. + private bool LegacySkyMotion() + { + SetPassContext("SkyMotion", PassFlags.None); + bool drawn = base.RenderOptimumSkyMotion(); + SetPassContext("Frame", PassFlags.AllowSplit); + return drawn; + } + + /// + /// Pass 3 - the SSAO pass, its bilateral blur and the AO composite - drawn natively + /// (VulkanClientPlatform.NativeSsao.cs). The lib virtual, which holds the OpenGL body's inline + /// code, is the old route: the differential tests compare the two, and a frame the native + /// route cannot draw (no device, no targets, no shader programs) falls back to it whole. + /// + private void PostStepAmbientOcclusion(float[] projectMatrix) + { + if (UseNativePostChain && NativeAmbientOcclusionReady()) + { + NativeAmbientOcclusion(projectMatrix); + return; + } + OptimumPostAmbientOcclusion(projectMatrix); + } + + /// + /// Pass 4, the TAA resolve. The lib body keeps the temporal contract - the guards, the + /// reset decision, the resolved textures and the history parity - and its draw seam is + /// on the native route. + /// + private bool PostStepTaaResolve() => RenderOptimumTaaResolve(); + + /// + /// The TAA sharpen runs at the blit boundary, after AfterFinalComposition overlays; its draw seam is + /// . The method remains a small route helper so the lib body + /// owns the skip conditions and the FSR no-double-sharpening rule. + /// + private int PostStepTaaSharpen(int resolvedScene) => RenderOptimumTaaSharpen(resolvedScene); + + /// Pass 6, the bloom chain, drawn natively (VulkanClientPlatform.NativePostFinal.cs). + private void PostStepBloom(int scene, int glow) => NativeBloom(scene, glow); + + /// Pass 7, god rays, drawn natively. + private void PostStepGodRays(int scene, int glow) => NativeGodRays(scene, glow); + + /// Pass 8, the FXAA luma prepass or the pass-through blit into Luma, drawn natively. + private void PostStepFxaaOrBlit(int scene) => NativePostLuma(scene); + + /// + /// The chain's epilogue: blending back on and Primary bound again. State, not a draw - it is + /// the GL-shaped handoff every stage after the chain inherits, so it stays as it is. + /// + private void PostStepFinish() => OptimumPostFinish(); + + /// LEGACY - pass 6 on the OpenGL body. replaces it; kept as the old route. + private void LegacyBloom(int scene, int glow) => OptimumPostBloom(scene, glow); + + /// LEGACY - pass 7 on the OpenGL body. replaces it; kept as the old route. + private void LegacyGodRays(int scene, int glow) => OptimumPostGodRays(scene, glow); + + /// LEGACY - pass 8 on the OpenGL body. replaces it; kept as the old route. + private void LegacyPostLuma(int scene) => OptimumPostLuma(scene); + + /// LEGACY - pass 9 on the OpenGL body. replaces it; kept as the old route. + private void LegacyFinalComposition() + { + SetPassContext("FinalComposition", PassFlags.None); + base.RenderFinalComposition(); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + // ------------------------------------------------------------------ shared plumbing + + /// + /// The guards and + /// open their window under, minus + /// the draw-buffer mask a native pass does not use: a native pass says which slots it + /// writes, so the window is the pass's colour slots. + /// + private bool NativeMotionAttachmentWritable(FrameBufferRef primary) + { + if (OptimumMotionWriteActive) return false; + if (MotionAttachmentIndex < 0 || !TaaTargetsReady) return false; + if (!OptimumConfig.EffectiveTaa) return false; + if (!OptimumTemporal.Frame.JitterActive) return false; + // The same "Primary is the target being drawn into" invariant, so a native pass never + // writes the motion attachment from under another target. + return ReferenceEquals(CurrentFrameBuffer, primary); + } + + /// Primary's default colour set: four attachments with the SSAO G-buffer, two without. + private uint NativeWorldColorSlots() => OptimumRenderSsao ? 0b1111u : 0b11u; + + private static int NativeSlotCount(uint slots) + { + int count = 0; + for (int i = 0; i < RenderLimits.MaxColorAttachments; i++) + { + if (((slots >> i) & 1) != 0) count = i + 1; + } + return count; + } + + /// + /// A pass that keeps the viewport, the way the OpenGL body's bind-only setter does: both + /// native passes here draw into the viewport the world stage left. + /// + private bool BeginNativeKeepViewportPass(string name, int framebufferId, uint colorSlots, int[] reads) + { + Rect2D viewport = StatedViewport(); + return device.BeginNativePass(new NativePassDescription + { + Name = name, + FramebufferId = framebufferId, + ColorSlots = colorSlots, + Reads = reads, + Flags = PassFlags.None, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + }); + } + + /// + /// A pass that owns its viewport, the way the OpenGL body's full CurrentFrameBuffer setter + /// does: bind the target and set the viewport to its own size. + /// + private bool BeginNativeTargetPass(string name, int framebufferId, uint colorSlots, + int width, int height, int[] reads) => + device.BeginNativePass(new NativePassDescription + { + Name = name, + FramebufferId = framebufferId, + ColorSlots = colorSlots, + Reads = reads, + Flags = PassFlags.None, + ViewportWidth = width, + ViewportHeight = height, + }); + + /// + /// The pipeline for one native pass of the chain, with the fixed state stated outright. The + /// device caches by (program, formats, blend, depth, cull, topology), so a pass whose blend + /// changes with the motion window simply gets the other pipeline, and the placements are + /// re-resolved whenever the pipeline object changes (a relink makes a new one). + /// + private NativePipeline? NativePostPipeline(NativeFullscreenPass pass, ShaderProgramBase program, + int framebufferId, uint colorSlots, AttachmentBlend[] blend, + bool depthTest, bool depthWrite, CompareOp depthCompare) + { + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, colorSlots); + if (formats == null) return null; + + NativePipeline? pipeline = device.RequestNativePipeline(new NativePipelineDescription + { + ProgramId = program.ProgramId, + PassName = pass.PassName, + Blend = blend, + DepthTest = depthTest, + DepthWrite = depthWrite, + DepthCompare = depthCompare, + Cull = CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + Targets = formats, + }, out string error); + + if (pipeline == null) + { + if (!pass.Reported) + { + pass.Reported = true; + Logger.Warning("Optimum: no native pipeline for '{0}': {1}", pass.PassName, error); + } + pass.Pipeline = null; + return null; + } + + pass.Reported = false; + if (!ReferenceEquals(pipeline, pass.Pipeline)) pass.Adopt(pipeline, formats); + return pipeline; + } + + /// A mat4 at its placement: sixteen floats, column-major, as the program declares it. + private void WriteNativeMatrix(NativePipeline pipeline, NativeUniform uniform, float[] values) => + WriteNativeFloats(pipeline, uniform, values); + + /// + /// A float array at its placement. The record and push blocks are scalar-packed, so a float[] + /// the game already holds - a matrix, or the SSAO sample kernel - copies straight in. + /// + private void WriteNativeFloats(NativePipeline pipeline, NativeUniform uniform, float[] values) => + device.WriteNative(pipeline, uniform, MemoryMarshal.AsBytes(new ReadOnlySpan(values))); +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostFinal.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostFinal.cs new file mode 100644 index 00000000..07720b88 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativePostFinal.cs @@ -0,0 +1,469 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), stage 1: the tail of the +// post chain - the bloom chain, god rays, the FXAA luma step and final composition - drawn +// through the native device API instead of the GL-shaped platform calls. +// +// Every pass here follows VulkanClientPlatform.NativeBlit.cs: a pipeline requested with its fixed +// state stated outright (per-attachment blend, depth test/write/compare, cull, topology, target +// formats), a pass declared with its explicit reads and colour slots, uniforms written by +// placement and sampled textures resolved straight to bindless slots. The values are the ones the +// OpenGL body computes, read from client state (decision 3): OptimumRenderBloom, +// OptimumRenderGodRays, OptimumRenderFxaa, OptimumSsaaLevel, OptimumRenderSsao, +// OptimumAmbientOcclusionTexture and OptimumSsaoInScene, plus ClientSettings and ShaderUniforms. +// +// The GL-shaped state calls the OpenGL body makes around these passes stay, outside every native +// pass: they are what the rest of the frame - the chain's epilogue, the GUI stage, a mod renderer +// - inherits, exactly as it does on the OpenGL body. +// +// Viewports: each of these targets is sized to the viewport its LoadFrameBuffer case sets (the +// FindBright and Luma cases set none and inherit the full render resolution, which is their own +// size), so every pass here draws into its whole target and the arithmetic cannot drift. +public partial class VulkanClientPlatform +{ + // ------------------------------------------------------------------ pass 6: bloom + + private readonly NativeFullscreenPass nativeFindBright = + new("findbright", new[] { "ambientBloomLevel", "extraBloom" }, new[] { "colorTex", "glowTex" }); + + // One instance per blur target: the program is the same, but a pipeline is per target + // formats, and each instance keeps its own resolved placements. + private readonly NativeFullscreenPass nativeBlurMedHorizontal = + new("blur", new[] { "frameSize", "isVertical" }, new[] { "inputTexture" }); + + private readonly NativeFullscreenPass nativeBlurMedVertical = + new("blur", new[] { "frameSize", "isVertical" }, new[] { "inputTexture" }); + + private readonly NativeFullscreenPass nativeBlurLowHorizontal = + new("blur", new[] { "frameSize", "isVertical" }, new[] { "inputTexture" }); + + private readonly NativeFullscreenPass nativeBlurLowVertical = + new("blur", new[] { "frameSize", "isVertical" }, new[] { "inputTexture" }); + + /// + /// The bloom chain, drawn natively: find-bright into the full-resolution target, then the + /// two blur ping-pongs at half and quarter resolution. The OpenGL body + /// (ClientPlatformWindows.OptimumPostBloom) turns blending off for the whole block, sets the + /// blur's frameSize exactly once - at the full resolution, before the half-resolution + /// pair, and never again for the quarter-resolution pair - and puts the viewport and + /// blending back at the end. All of that is reproduced here, the stale frameSize + /// included: it is what the vanilla image is made of, not a bug to fix. + /// + private void NativeBloom(int scene, int glow) + { + if (!OptimumRenderBloom) return; + + List buffers = FrameBuffers; + ShaderProgramFindbright findbright = ShaderPrograms.Findbright; + ShaderProgramBlur blur = ShaderPrograms.Blur; + FrameBufferRef? findBrightTarget = NativePostTarget(buffers, FindBrightIndex); + FrameBufferRef? medHorizontal = NativePostTarget(buffers, BlurHorizontalMedResIndex); + FrameBufferRef? medVertical = NativePostTarget(buffers, BlurVerticalMedResIndex); + FrameBufferRef? lowHorizontal = NativePostTarget(buffers, BlurHorizontalLowResIndex); + FrameBufferRef? lowVertical = NativePostTarget(buffers, BlurVerticalLowResIndex); + + if (!NativeProgramUsable(findbright) || !NativeProgramUsable(blur) || + findBrightTarget == null || medHorizontal == null || medVertical == null || + lowHorizontal == null || lowVertical == null) + { + LegacyBloom(scene, glow); + return; + } + + NativePipeline? bright = NativePipelineFor(nativeFindBright, findbright, findBrightTarget.FboId); + NativePipeline? medH = NativePipelineFor(nativeBlurMedHorizontal, blur, medHorizontal.FboId); + NativePipeline? medV = NativePipelineFor(nativeBlurMedVertical, blur, medVertical.FboId); + NativePipeline? lowH = NativePipelineFor(nativeBlurLowHorizontal, blur, lowHorizontal.FboId); + NativePipeline? lowV = NativePipelineFor(nativeBlurLowVertical, blur, lowVertical.FboId); + if (bright == null || medH == null || medV == null || lowH == null || lowV == null) + { + LegacyBloom(scene, glow); + return; + } + + Size2i client = OptimumWindowClientSize(); + float ssaa = OptimumSsaaLevel; + + // The block's blend state, left where the OpenGL body leaves it, outside the passes. + GlToggleBlend(on: false); + + if (BeginNativePostPass("Post/" + FindBrightIndex, findBrightTarget.FboId, new[] { scene, glow }, + transient: true)) + { + device.WriteNative(bright, nativeFindBright.Uniforms[0], NativeAmbientBloomLevel()); + device.WriteNative(bright, nativeFindBright.Uniforms[1], ShaderUniforms.ExtraBloom); + device.DrawNativeFullscreen(bright, new[] + { + new NativeTexture(nativeFindBright.Samplers[0], scene), + new NativeTexture(nativeFindBright.Samplers[1], glow), + }); + } + device.EndNativePass(); + + // frameSize is the full-resolution value the OpenGL body sets once here and reuses for + // all four blur draws, including the two that write a quarter-resolution target. + float blurWidth = client.Width * ssaa; + float blurHeight = client.Height * ssaa; + + NativeBlurStep(medH, nativeBlurMedHorizontal, "Post/" + BlurHorizontalMedResIndex, + medHorizontal.FboId, findBrightTarget.ColorTextureIds[0], vertical: 0, blurWidth, blurHeight); + NativeBlurStep(medV, nativeBlurMedVertical, "Post/" + BlurVerticalMedResIndex, + medVertical.FboId, medHorizontal.ColorTextureIds[0], vertical: 1, blurWidth, blurHeight); + NativeBlurStep(lowH, nativeBlurLowHorizontal, "Post/" + BlurHorizontalLowResIndex, + lowHorizontal.FboId, medVertical.ColorTextureIds[0], vertical: 0, blurWidth, blurHeight); + NativeBlurStep(lowV, nativeBlurLowVertical, "Post/" + BlurVerticalLowResIndex, + lowVertical.FboId, lowHorizontal.ColorTextureIds[0], vertical: 1, blurWidth, blurHeight); + + // What the rest of the frame inherits from this block on the OpenGL body. + GlViewport(0, 0, (int)(ssaa * client.Width), (int)(ssaa * client.Height)); + GlToggleBlend(on: true); + } + + private void NativeBlurStep(NativePipeline pipeline, NativeFullscreenPass pass, string name, + int framebufferId, int input, int vertical, float frameWidth, float frameHeight) + { + if (BeginNativePostPass(name, framebufferId, new[] { input }, transient: true)) + { + device.WriteNative(pipeline, pass.Uniforms[0], frameWidth, frameHeight); + device.WriteNative(pipeline, pass.Uniforms[1], vertical); + device.DrawNativeFullscreen(pipeline, new[] { new NativeTexture(pass.Samplers[0], input) }); + } + device.EndNativePass(); + } + + // --------------------------------------------------------------- pass 7: god rays + + private readonly NativeFullscreenPass nativeGodRays = new("godrays", + new[] + { + "invFrameSizeIn", "maxGodRaySamples", "sunPosScreenIn", "sunPos3dIn", + "playerViewVector", "dusk", "iGlobalTimeIn", + }, + new[] { "inputTexture", "glowParts" }); + + /// + /// God rays, drawn natively into the half-resolution target. The OpenGL body + /// (ClientPlatformWindows.OptimumPostGodRays) toggles no blend of its own, so the draw runs + /// with whatever the steps before it left - blending on in the source-alpha mode, which the + /// shader's outColor.a = 1 makes indistinguishable from blending off. The pipeline + /// states that mode outright rather than inheriting a tracked one, and the viewport reset the + /// body ends with stays, outside the pass. + /// + /// sunPos3dIn comes from ShaderUniforms.LightPosition3D here and from + /// SunPosition3D in the final composition: two different fields behind one uniform + /// name, as on the OpenGL body. + /// + private void NativeGodRays(int scene, int glow) + { + if (!OptimumRenderGodRays) return; + + List buffers = FrameBuffers; + ShaderProgramGodrays godrays = ShaderPrograms.Godrays; + FrameBufferRef? target = NativePostTarget(buffers, GodRaysIndex); + if (!NativeProgramUsable(godrays) || target == null) + { + LegacyGodRays(scene, glow); + return; + } + + NativePipeline? pipeline = NativePostPipeline(nativeGodRays, godrays, target.FboId, 1u, + NativeStandardBlend(), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) + { + LegacyGodRays(scene, glow); + return; + } + + Size2i client = OptimumWindowClientSize(); + float ssaa = OptimumSsaaLevel; + + if (BeginNativePostPass("Post/" + GodRaysIndex, target.FboId, new[] { scene, glow }, transient: false)) + { + // The input texel size is the full-resolution one, describing the texture the pass + // samples and not the half-resolution target it writes. + device.WriteNative(pipeline, nativeGodRays.Uniforms[0], + 1f / (client.Width * ssaa), 1f / (client.Height * ssaa)); + device.WriteNative(pipeline, nativeGodRays.Uniforms[1], OptimumConfig.GodRaysSampleLimit); + WriteNativeVec3(pipeline, nativeGodRays.Uniforms[2], ShaderUniforms.SunPositionScreen); + WriteNativeVec3(pipeline, nativeGodRays.Uniforms[3], ShaderUniforms.LightPosition3D); + WriteNativeVec3(pipeline, nativeGodRays.Uniforms[4], ShaderUniforms.PlayerViewVector); + device.WriteNative(pipeline, nativeGodRays.Uniforms[5], ShaderUniforms.Dusk); + device.WriteNative(pipeline, nativeGodRays.Uniforms[6], (float)EllapsedMs / 1000f); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeGodRays.Samplers[0], scene), + new NativeTexture(nativeGodRays.Samplers[1], glow), + }); + } + device.EndNativePass(); + + GlViewport(0, 0, (int)(ssaa * client.Width), (int)(ssaa * client.Height)); + } + + // -------------------------------------------------------- pass 8: FXAA luma or blit + + private readonly NativeFullscreenPass nativeLuma = + new("luma", Array.Empty(), new[] { "scene" }); + + private readonly NativeFullscreenPass nativeLumaBlit = + new("blit", Array.Empty(), new[] { "scene" }); + + /// + /// The Luma step, drawn natively. The OpenGL body (ClientPlatformWindows.OptimumPostLuma) + /// picks the branch on RenderFXAA && !TaaResolvedThisFrame: the FXAA luma + /// prepass over the raw jittered Primary colour - bypassing the whole resolved chain - or a + /// pass-through blit of the chain's scene. Both go into the Luma target, and the Luma case of + /// LoadFrameBuffer turns blending off without putting it back; the chain's epilogue + /// does that. + /// + private void NativePostLuma(int scene) + { + List buffers = FrameBuffers; + FrameBufferRef? target = NativePostTarget(buffers, LumaIndex); + FrameBufferRef? primary = NativePostTarget(buffers, PrimaryIndex); + bool fxaa = OptimumRenderFxaa && !TaaResolvedThisFrame; + if (fxaa && (primary == null || primary.ColorTextureIds == null || primary.ColorTextureIds.Length < 1)) + { + LegacyPostLuma(scene); + return; + } + + ShaderProgramBase program = fxaa ? ShaderPrograms.Luma : ShaderPrograms.Blit; + NativeFullscreenPass pass = fxaa ? nativeLuma : nativeLumaBlit; + if (!NativeProgramUsable(program) || target == null) + { + LegacyPostLuma(scene); + return; + } + + NativePipeline? pipeline = NativePipelineFor(pass, program, target.FboId); + if (pipeline == null) + { + LegacyPostLuma(scene); + return; + } + + int source = fxaa ? primary!.ColorTextureIds[0] : scene; + + // The Luma case of LoadFrameBuffer turns blending off and leaves it off. + SetBlendEnabled(false); + if (BeginNativePostPass("Post/" + LumaIndex, target.FboId, new[] { source }, transient: false)) + { + device.DrawNativeFullscreen(pipeline, new[] { new NativeTexture(pass.Samplers[0], source) }); + } + device.EndNativePass(); + } + + // ------------------------------------------------------- pass 9: final composition + + private readonly NativeFullscreenPass nativeFinal = new("final", + new[] + { + "ambientBloomLevel", "optimumSsaoInScene", "optimumAoDebug", "invFrameSizeIn", + "gammaLevel", "extraGamma", "contrastLevel", "brightnessLevel", "sepiaLevel", + "windWaveCounter", "glitchEffectStrength", "sunPosScreenIn", "sunPos3dIn", + "playerViewVector", "damageVignetting", "damageVignettingSide", "frostVignetting", + }, + new[] { "primaryScene", "glowParts", "bloomParts", "godrayParts", "ssaoScene" }); + + /// + /// The final composition, drawn natively. This is the attachment-subset pass of the chain: it + /// writes Primary colour 0 while sampling Primary colour 1 as the glow on a frame the TAA + /// resolve did not run, so the pass declares every bound colour slot except 1 and slot 1 + /// leaves the scope for the pass - one barrier out to the shader-read layout and one back + /// when the next pass attaches it, and no feedback copy. + /// + /// The values are the OpenGL body's (ClientPlatformWindows.RenderFinalComposition), including + /// the two it writes unconditionally: optimumSsaoInScene is written on every frame - + /// a declared uniform left unset reads back as whatever the uniform ring last held - and the + /// bloom and god-ray inputs are sampled whether or not their passes ran this frame. + /// + private void NativeFinalComposition() + { + if (!offscreenBufferActive) return; + + List buffers = FrameBuffers; + ShaderProgramFinal final = ShaderPrograms.Final; + FrameBufferRef? primary = NativePostTarget(buffers, PrimaryIndex); + FrameBufferRef? luma = NativePostTarget(buffers, LumaIndex); + FrameBufferRef? bloom = NativePostTarget(buffers, BlurVerticalLowResIndex); + FrameBufferRef? godRays = NativePostTarget(buffers, GodRaysIndex); + if (!NativeProgramUsable(final) || primary == null || luma == null || bloom == null || godRays == null || + primary.ColorTextureIds == null || primary.ColorTextureIds.Length < 2) + { + LegacyFinalComposition(); + return; + } + + bool renderSsao = OptimumRenderSsao; + bool aoDebugView = OptimumConfig.AmbientOcclusionDebugView && renderSsao; + int aoTexture = OptimumAmbientOcclusionTexture; + FrameBufferRef? ssaoBlur = NativePostTarget(buffers, SsaoBlurVerticalIndex); + int ssaoScene = 0; + if (renderSsao) + { + ssaoScene = aoDebugView && aoTexture != 0 + ? aoTexture + : ssaoBlur?.ColorTextureIds != null && ssaoBlur.ColorTextureIds.Length > 0 + ? ssaoBlur.ColorTextureIds[0] + : 0; + } + + // Primary colour 1 stays out of the pass so the glow can be sampled from it. + const uint slots = ~(1u << 1); + + // The draw-buffer selection and the blend the OpenGL body sets around the pass: outside + // it, so everything after the composition inherits what it always did. + BeginFinalCompositionDrawBuffers(); + GlToggleBlend(on: true); + + NativePipeline? pipeline = NativePostPipeline(nativeFinal, final, primary.FboId, slots, + NativeFinalBlend(slots), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) + { + RestoreWorldDrawBuffers(renderSsao); + LegacyFinalComposition(); + return; + } + + int primaryScene = luma.ColorTextureIds[0]; + int glow = OptimumPostGlowTexture(); + int bloomParts = bloom.ColorTextureIds[0]; + int godrayParts = godRays.ColorTextureIds[0]; + + var reads = new List { primaryScene, glow, bloomParts, godrayParts }; + if (ssaoScene > 0 && !reads.Contains(ssaoScene)) reads.Add(ssaoScene); + + Size2i client = OptimumWindowClientSize(); + float ssaa = OptimumSsaaLevel; + + SetPassContext("FinalComposition", PassFlags.None); + if (device.BeginNativePass(new NativePassDescription + { + Name = "FinalComposition/0", + FramebufferId = primary.FboId, + ColorSlots = slots, + Reads = reads.ToArray(), + Flags = PassFlags.None, + })) + { + NativeUniform[] u = nativeFinal.Uniforms; + device.WriteNative(pipeline, u[0], NativeAmbientBloomLevel()); + device.WriteNative(pipeline, u[1], (OptimumSsaoInScene || !renderSsao) ? 1 : 0); + device.WriteNative(pipeline, u[2], aoDebugView ? 1 : 0); + device.WriteNative(pipeline, u[3], 1f / (client.Width * ssaa), 1f / (client.Height * ssaa)); + device.WriteNative(pipeline, u[4], ClientSettings.GammaLevel); + device.WriteNative(pipeline, u[5], ClientSettings.ExtraGammaLevel); + device.WriteNative(pipeline, u[6], ShaderUniforms.ExtraContrastLevel); + device.WriteNative(pipeline, u[7], ClientSettings.BrightnessLevel + + Math.Max(0f, ShaderUniforms.DropShadowIntensity * 2f - 1.66f) / 3f); + device.WriteNative(pipeline, u[8], ShaderUniforms.SepiaLevel + ShaderUniforms.ExtraSepia); + device.WriteNative(pipeline, u[9], ShaderUniforms.WindWaveCounter); + device.WriteNative(pipeline, u[10], ShaderUniforms.GlitchStrength); + if (OptimumRenderGodRays) + { + WriteNativeVec3(pipeline, u[11], ShaderUniforms.SunPositionScreen); + WriteNativeVec3(pipeline, u[12], ShaderUniforms.SunPosition3D); + WriteNativeVec3(pipeline, u[13], ShaderUniforms.PlayerViewVector); + } + device.WriteNative(pipeline, u[14], ShaderUniforms.DamageVignetting); + device.WriteNative(pipeline, u[15], ShaderUniforms.DamageVignettingSide); + device.WriteNative(pipeline, u[16], ShaderUniforms.FrostVignetting); + + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeFinal.Samplers[0], primaryScene), + new NativeTexture(nativeFinal.Samplers[1], glow), + new NativeTexture(nativeFinal.Samplers[2], bloomParts), + new NativeTexture(nativeFinal.Samplers[3], godrayParts), + new NativeTexture(nativeFinal.Samplers[4], ssaoScene), + }); + } + device.EndNativePass(); + + RestoreWorldDrawBuffers(renderSsao); + SetPassContext("Frame", PassFlags.AllowSplit); + } + + // ------------------------------------------------------------------ shared plumbing + + /// + /// The bloom level the find-bright pass and the final composition both take, bit for bit the + /// same expression on both (ClientPlatformWindows). + /// + private float NativeAmbientBloomLevel() => + ClientSettings.AmbientBloomLevel / 100f + ShaderUniforms.AmbientBloomLevelAdd[0] + + ShaderUniforms.AmbientBloomLevelAdd[1] + ShaderUniforms.AmbientBloomLevelAdd[2] + + ShaderUniforms.AmbientBloomLevelAdd[3]; + + /// A vec3 at its placement. + private void WriteNativeVec3(NativePipeline pipeline, NativeUniform uniform, Vec3f value) => + device.WriteNative(pipeline, uniform, value.X, value.Y, value.Z); + + /// A post target of the chain, or null when the frame buffers do not hold it. + private static FrameBufferRef? NativePostTarget(List buffers, int index) + { + if (buffers == null || index < 0 || index >= buffers.Count) return null; + FrameBufferRef target = buffers[index]; + if (target == null || target.Disposed || target.FboId == 0) return null; + if (target.ColorTextureIds == null || target.ColorTextureIds.Length == 0) return null; + return target; + } + + private static bool NativeProgramUsable(ShaderProgramBase? program) => + program != null && !program.LoadError && !program.Disposed && program.ProgramId > 0; + + /// + /// One colour-0 pass of the chain's tail, drawing into the whole target - which is the + /// viewport its LoadFrameBuffer case sets on the OpenGL body. + /// + private bool BeginNativePostPass(string name, int framebufferId, int[] reads, bool transient) => + device.BeginNativePass(new NativePassDescription + { + Name = name, + FramebufferId = framebufferId, + ColorSlots = 1u, + Reads = reads, + TransientSlots = transient ? 1u : 0u, + Flags = PassFlags.None, + }); + + /// Blending on in the source-alpha mode on colour 0, as GlToggleBlend(true) sets it. + private static AttachmentBlend[] NativeStandardBlend() + { + AttachmentBlend blend = AttachmentBlend.Default; + blend.Enabled = true; + return new[] { blend }; + } + + /// + /// The final composition's blend: the source-alpha mode on colour 0 and no write anywhere + /// else, so the G-buffer and motion attachments the pass keeps in its scope come out of it + /// exactly as they went in. + /// + private static AttachmentBlend[] NativeFinalBlend(uint slots) + { + var blend = new AttachmentBlend[NativeSlotCount(slots)]; + for (int i = 0; i < blend.Length; i++) + { + if (i == 0) + { + blend[0] = AttachmentBlend.Default; + blend[0].Enabled = true; + continue; + } + blend[i].WriteMask = 0; + } + return blend; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSky.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSky.cs new file mode 100644 index 00000000..9f0e0cdf --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSky.cs @@ -0,0 +1,449 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b decision 5 +// stage 2: the first world system on the native device API, and the proof that its mesh-draw +// entry points work end to end. +// +// What it draws: the sky dome - the 250-unit icosahedron SystemRenderSkyColor renders the sky +// gradient onto, once per frame, first in the Opaque stage. +// Where the other side is: ClientPlatformAbstract.RenderSkyDome's neutral body, which is the +// RenderMesh(MeshRef) call this seam replaced and which the OpenGL path still runs +// (ClientPlatformWindows.RenderMesh -> GL.DrawElements). NativeSkyEnabled false takes that +// route on the Vulkan device too, which is what the differential test compares against. +// Target and slots: the framebuffer the Opaque stage bound (Primary), every bound colour slot +// in the pass so the scope is the one the emulated draw opens; sky.frag writes outColor at 0 +// and outGlow at 1 and the pipeline masks every other slot off, so the motion attachment and +// the G-buffer slots keep their contents exactly as they do on GL. +// State that is not obvious: +// - no depth test and no depth write: the caller has already called GlDisableDepthTest, and +// the dome is meant to sit behind everything; +// - no culling: the OpenGL body draws with whatever cull state the stage before it left (a +// shadow pass leaves it off, no shadow pass leaves back-face culling on), and the dome is +// a closed hull drawn without depth test, so None is the state that draws every triangle +// the GL path can draw; +// - blend disabled: sky.frag writes alpha 1 into both its outputs, so the blend the stage +// happens to have on makes no difference to the result; +// - the two textures are passed as handles, because a native pass resolves what it samples +// from handles rather than from the texture units the program's setters bound them to. +// What pins it: NativeSkyTests (old route against native route, pixels) and +// Optimum.Tests/native-sky-coverage-tests.cs (the lib seam). +public partial class VulkanClientPlatform +{ + /// + /// False runs the seam's neutral body - the OpenGL body's RenderMesh - on the Vulkan device + /// instead of the native pass: the old route the differential test compares against, in the + /// pattern of . + /// + internal bool NativeSkyEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_SKY") != "0"; + + /// + /// The sky program's pipeline and the placements its draw writes through. "modelViewMatrix" + /// is the one value written per draw; everything else the client system set through the + /// program's own setters is already in the program's record shadow when the draw binds set 2. + /// + private readonly NativeMeshPass nativeSky = + new("sky", new[] { "modelViewMatrix" }, new[] { "sky", "glow" }); + + /// + /// A native program drawn with a mesh: its pipeline for one target and one mesh shape, and + /// the placements its draws write through. The fullscreen twin is NativeFullscreenPass in + /// VulkanClientPlatform.NativeBlit.cs; this one also keys on the mesh's vertex layout, + /// because that is part of the pipeline. + /// + private sealed class NativeMeshPass + { + public NativeMeshPass(string passName, string[] uniforms, string[] samplers) + { + PassName = passName; + UniformNames = uniforms; + SamplerNames = samplers; + Uniforms = new NativeUniform[uniforms.Length]; + Samplers = new NativeSamplerSlot[samplers.Length]; + } + + public string PassName { get; } + public string[] UniformNames { get; } + public string[] SamplerNames { get; } + + /// Resolved once per pipeline, never by name per draw. + public NativeUniform[] Uniforms; + public NativeSamplerSlot[] Samplers; + + public NativePipeline? Pipeline; + public RenderTargetFormats? Formats; + public int LayoutId = -1; + public bool Reported; + + public void Adopt(NativePipeline pipeline, RenderTargetFormats formats, int layoutId) + { + Pipeline = pipeline; + Formats = formats; + LayoutId = layoutId; + for (int i = 0; i < UniformNames.Length; i++) Uniforms[i] = pipeline.Uniform(UniformNames[i]); + for (int i = 0; i < SamplerNames.Length; i++) Samplers[i] = pipeline.Sampler(SamplerNames[i]); + } + } + + /// + /// The pipeline for one mesh program against one target and one mesh shape, rebuilt only + /// when the program was relinked, the target's formats changed or the mesh's layout did. + /// + private NativePipeline? NativeMeshPipelineFor(NativeMeshPass pass, ShaderProgramBase program, int framebufferId, + uint colorSlots, int layoutId, NativePipelineDescription description) + { + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, colorSlots); + if (formats == null) return null; + + if (pass.Pipeline != null && pass.Pipeline.ProgramId == program.ProgramId && + formats.Equals(pass.Formats) && pass.LayoutId == layoutId && + SameFixedState(pass.Pipeline.Description, description) && device.IsNativePipelineLive(pass.Pipeline)) + { + return pass.Pipeline; + } + + description.ProgramId = program.ProgramId; + description.PassName = pass.PassName; + description.VertexLayoutId = layoutId; + description.Targets = formats; + + NativePipeline? pipeline = device.RequestNativePipeline(description, out string error); + if (pipeline == null) + { + if (!pass.Reported) + { + pass.Reported = true; + Logger.Warning("Optimum: no native pipeline for '{0}': {1}", pass.PassName, error); + } + pass.Pipeline = null; + return null; + } + + pass.Reported = false; + pass.Adopt(pipeline, formats, layoutId); + return pipeline; + } + + /// + /// Whether two descriptions ask for the same fixed state, which is what makes the cached + /// pipeline of a usable for the next draw through it. + /// + /// Without this the one-entry cache answered any request for the same program, target and + /// mesh shape with the pipeline it happened to build first: the aiming reticle's 0.5 and + /// 1.0 line widths would then both rasterize at whichever came first, and a system that + /// turns blending on and off between draws would blend both or neither. The device's own + /// table keys on all of it (NativePipelineCacheKey), so falling through to + /// costs a dictionary lookup, not a + /// pipeline. + /// + private static bool SameFixedState(NativePipelineDescription cached, NativePipelineDescription wanted) + { + if (cached.DepthTest != wanted.DepthTest || cached.DepthWrite != wanted.DepthWrite || + cached.DepthCompare != wanted.DepthCompare || cached.Cull != wanted.Cull || + cached.FrontFace != wanted.FrontFace || cached.Topology != wanted.Topology || + cached.PolygonMode != wanted.PolygonMode || cached.SamplesBoundDepth != wanted.SamplesBoundDepth || + !cached.LineWidth.Equals(wanted.LineWidth) || cached.Blend.Length != wanted.Blend.Length) + { + return false; + } + for (int i = 0; i < cached.Blend.Length; i++) + { + if (!cached.Blend[i].Equals(wanted.Blend[i])) return false; + } + return true; + } + + /// Every bound colour slot of a target: the scope the emulated draw would open. + private static uint NativeAllColorSlots(FrameBufferRef target) + { + int count = target.ColorTextureIds?.Length ?? 0; + return count >= 32 ? uint.MaxValue : (1u << count) - 1u; + } + + /// + /// Opaque-stage fixed state for a world pass that draws over everything: no depth, no + /// culling, no blending, one entry per colour attachment so the slots the program does not + /// write are masked off rather than left to Vulkan's undefined contents (rule 9). + /// + private static AttachmentBlend[] OpaqueSlots(RenderTargetFormats formats) + { + var blend = new AttachmentBlend[Math.Max(formats.ColorFormats.Length, 1)]; + for (int i = 0; i < blend.Length; i++) blend[i] = AttachmentBlend.Default; + return blend; + } + + /// The sky dome's draw: the native pass, or the neutral body's RenderMesh. + public override void RenderSkyDome(MeshRef skyDome, int skyTextureId, int glowTextureId, float[] modelViewMatrix) + { + if (!NativeSkyEnabled || device == null || skyDome == null) + { + base.RenderSkyDome(skyDome!, skyTextureId, glowTextureId, modelViewMatrix); + return; + } + + FrameBufferRef target = CurrentFrameBuffer; + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + var vao = skyDome as VAO; + if (target == null || program == null || vao == null || vao.VaoId == 0 || vao.Disposed) + { + base.RenderSkyDome(skyDome, skyTextureId, glowTextureId, modelViewMatrix); + return; + } + + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) + { + base.RenderSkyDome(skyDome, skyTextureId, glowTextureId, modelViewMatrix); + return; + } + + uint slots = NativeAllColorSlots(target); + RenderTargetFormats? formats = device.NativeTargetFormats(target.FboId, slots); + if (formats == null) + { + base.RenderSkyDome(skyDome, skyTextureId, glowTextureId, modelViewMatrix); + return; + } + + NativePipeline? pipeline = NativeMeshPipelineFor(nativeSky, program, target.FboId, slots, layoutId, + new NativePipelineDescription + { + Blend = OpaqueSlots(formats), + DepthTest = false, + DepthWrite = false, + Cull = CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + }); + if (pipeline == null) + { + base.RenderSkyDome(skyDome, skyTextureId, glowTextureId, modelViewMatrix); + return; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + Rect2D viewport = StatedViewport(); + if (device.BeginNativePass(new NativePassDescription + { + Name = "Sky/" + target.FboId, + FramebufferId = target.FboId, + ColorSlots = slots, + Reads = new[] { skyTextureId, glowTextureId }, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + })) + { + // The one per-draw write: the dome follows the player, so the model-view matrix is + // this draw's and nothing else's. Everything else the system set is already in the + // program's record, which the draw snapshots into this frame's uniform ring. + if (modelViewMatrix != null && modelViewMatrix.Length >= 16) + { + device.WriteNative(pipeline, nativeSky.Uniforms[0], modelViewMatrix.AsSpan(0, 16)); + } + device.DrawNativeMesh(pipeline, vao.VaoId, new[] + { + new NativeTexture(nativeSky.Samplers[0], skyTextureId), + new NativeTexture(nativeSky.Samplers[1], glowTextureId), + }); + } + device.EndNativePass(); + + SetPassContext(outer, outerFlags); + } +} + +// The cloud renderers of the forked VSEssentials (Systems/Weather/Newclouds), drawn natively. +// +// Both renderers end in a plain capi.Render.RenderMesh(quad), which lands in RenderMesh here; +// their state goes through OptimumForkGraphics, whose Vulkan implementation (VulkanForkGraphics) +// records it on this platform next to forwarding it to the device. The OpenGL side is the same +// fork code's GL branch plus ClientPlatformWindows.RenderMesh. +// +// - cloudmap (CloudRendererMap.OnRenderFrame): a fullscreen quad into the renderer's own device +// framebuffer (tile and colour attachments), bound by id through the fork surface, blending +// and depth test off. The target has no FrameBufferRef; the pass names the device id. +// - cloudvolumetric (CloudRendererVolumetric.OnRenderFrame, OIT stage): a fullscreen quad onto +// the Transparent target under the OIT contract, depth test off, sampling Primary's depth - +// which the Transparent target borrows as its own depth attachment, so the pipeline declares +// SamplesBoundDepth and writes no depth (GL writes none with the depth test off either). +// +// Pinned by Optimum.Tests/native-world-systems-coverage-tests.cs. +public partial class VulkanClientPlatform +{ + /// False sends both cloud draws to the generic stated route (OPTIMUM_VK_NATIVE_CLOUDS=0). + internal bool NativeCloudsEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_CLOUDS") != "0"; + + private readonly NativeMeshPass nativeCloudMap = + new("cloudmap", Array.Empty(), Array.Empty()); + + private readonly NativeMeshPass nativeCloudVolumetric = + new("cloudvolumetric", Array.Empty(), Array.Empty()); + + /// + /// The device framebuffer a fork renderer bound through OptimumForkGraphics.BindFramebuffer + /// and has not unbound yet; 0 when the fork's binding is the platform's current target again + /// (its restore) or the default framebuffer. + /// + private int forkFramebuffer; + + internal void NoteForkFramebuffer(int framebufferId) + { + FrameBufferRef current = CurrentFrameBuffer; + forkFramebuffer = current != null && current.FboId == framebufferId ? 0 : framebufferId; + } + + internal void NoteForkDepthTest(bool enabled) + { + statedDepthTest = enabled; + stated.DepthTest = enabled; + } + + internal void NoteForkBlend(bool enabled) + { + statedBlendOn = enabled; + stated.SetBlendEnabled(enabled); + } + + /// A cloud renderer's RenderMesh: the native pass, or false for the generic stated draw. + private bool TryRenderCloudsNative(MeshRef mesh) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeWorldEnabled || !NativeCloudsEnabled || device == null || mesh == null || program == null) + { + return false; + } + + string? name = program.PassName; + if (name != "cloudmap" && name != "cloudvolumetric") return false; + // The program the registry holds under that name, which is the fork's own registration. + if (!IsRegistryProgram(program)) return false; + + return name == "cloudmap" ? DrawCloudMapNative(program, mesh) : DrawCloudVolumetricNative(program, mesh); + } + + private bool DrawCloudMapNative(ShaderProgramBase program, MeshRef mesh) + { + var vao = mesh as VAO; + int framebufferId = forkFramebuffer; + if (vao == null || vao.VaoId == 0 || vao.Disposed || framebufferId <= 0) return false; + + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) return false; + // The attachments the fork's SetDrawBuffers enabled decide the scope's formats. + RenderTargetFormats? all = device.NativeTargetFormats(framebufferId, uint.MaxValue); + if (all == null || all.ColorFormats.Length == 0) return false; + uint slots = all.ColorFormats.Length >= 32 ? uint.MaxValue : (1u << all.ColorFormats.Length) - 1u; + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, slots); + if (formats == null) return false; + + NativePipeline? pipeline = NativeMeshPipelineFor(nativeCloudMap, program, framebufferId, slots, layoutId, + new NativePipelineDescription + { + Blend = OpaqueSlots(formats), + DepthTest = false, + DepthWrite = false, + Cull = CullModeFlags.None, + Topology = device.NativeMeshTopology(vao.VaoId), + }); + if (pipeline == null) return false; + + ResolveDeclaredSamplers(program, pipeline, out NativeTexture[] textures, out int[] reads); + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + Rect2D viewport = StatedViewport(); + bool drawn = false; + if (device.BeginNativePass(new NativePassDescription + { + Name = "CloudMap/" + framebufferId, + FramebufferId = framebufferId, + ColorSlots = slots, + Reads = reads, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + })) + { + drawn = device.DrawNativeMesh(pipeline, vao.VaoId, textures); + } + device.EndNativePass(); + SetPassContext(outer, outerFlags); + return drawn; + } + + private bool DrawCloudVolumetricNative(ShaderProgramBase program, MeshRef mesh) + { + FrameBufferRef bound = CurrentFrameBuffer; + if (forkFramebuffer != 0 || bound == null || nativeTransparentBlend == null || !IsTransparentTarget(bound) || + statedDepthTest) + { + return false; + } + + if (!NativeWorldPrepare(nativeCloudVolumetric, mesh, blending: statedBlendOn, depth: false, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline, + count => StatedWorldBlend(bound, count), depthWrite: false, samplesBoundDepth: true)) + { + return false; + } + + ResolveDeclaredSamplers(program, pipeline, out NativeTexture[] textures, out int[] reads); + // liquidDepth is not bound through the program: SystemRenderOITLayers points the sampler + // at unit 4 by value, and the GL draw reads whatever the liquid pass left there - the + // LiquidDepth target's depth texture. The native draw names that texture, since a + // frame-bound sampler resolved to 0 would replace the frame's liquid depth with a + // placeholder and cut every cloud off at the near plane. + string[] names = pipeline.SamplerNames; + for (int i = 0; i < names.Length; i++) + { + if (names[i] != "liquidDepth" || textures[i].TextureId != 0) continue; + FrameBufferRef? liquid = FrameBuffers is { Count: > (int)EnumFrameBuffer.LiquidDepth } + ? FrameBuffers[(int)EnumFrameBuffer.LiquidDepth] + : null; + if (liquid == null) return false; + textures[i] = new NativeTexture(textures[i].Sampler, liquid.DepthTextureId); + reads[i] = liquid.DepthTextureId; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + bool drawn = false; + if (NativeWorldBeginPass("CloudVolumetric", target, slots, reads)) + { + drawn = device.DrawNativeMesh(pipeline, vao.VaoId, textures); + } + NativeWorldEndPass(target, outer, outerFlags); + return drawn; + } + + /// Every sampler the pipeline declares, resolved from the program's declared textures. + private void ResolveDeclaredSamplers(ShaderProgramBase program, NativePipeline pipeline, + out NativeTexture[] textures, out int[] reads) + { + string[] names = pipeline.SamplerNames; + textures = new NativeTexture[names.Length]; + reads = new int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int id = DeclaredProgramTexture(program.ProgramId, names[i]); + textures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + reads[i] = id; + } + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSsao.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSsao.cs new file mode 100644 index 00000000..084850ed --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeSsao.cs @@ -0,0 +1,455 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Vintagestory.API.MathTools; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; +using Optimum.Render.Vulkan.AmbientOcclusion; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b stage 1c: the +// post chain's ambient-occlusion step drawn natively - the vanilla SSAO pass, its bilateral blur +// ping-pong and the AO composite that multiplies the visibility into the scene before the TAA +// resolve reads it. +// +// The step is the OpenGL body's (ClientPlatformWindows.OptimumPostAmbientOcclusion and the +// private ApplyOptimumSceneSsao it calls), value for value: the same guards, the same order, the +// same uniform expressions, the same textures. What changes is how each draw reaches the GPU - +// a pipeline built for stated fixed state (per-attachment blend, depth test/write/compare, cull, +// topology and the target's formats) instead of whatever the GL state tracker happens to hold, +// a pass that names its target, its written colour slots and the textures it samples instead of +// a draw-buffer mask, uniforms written by resolved placement instead of by name, and sampled +// textures resolved straight to bindless slots. +// +// Both AO modes run here. Vanilla SSAO renders into frameBuffers[13], is blurred through 15/14 +// and composed from 14; Optimum's GTAO (GtaoRenderer, already native and untouched by this file) +// hands the composite its visibility texture instead and the composite takes the OPTIMUMAO branch +// with optimumAoMode = 1. Either way the step ends by recording that AO is in the scene, which is +// what keeps the final composition from applying it a second time. +public partial class VulkanClientPlatform +{ + private readonly NativeFullscreenPass nativeSsao = new("ssao", + new[] { "screenSize", "projection", "samples", "temporalFrameIndex" }, + new[] { "gPosition", "gNormal", "texNoise", "revealage" }); + + private readonly NativeFullscreenPass nativeBilateralBlur = new("bilateralblur", + new[] { "frameSize", "isVertical" }, + new[] { "inputTexture", "depthTexture" }); + + private readonly NativeFullscreenPass nativeSceneSsao = new("scene-ssao", + new[] { "invRenderHeight", "optimumAoMode" }, + new[] { "ssaoScene", "gPositionScene", "revealageScene" }); + + /// The frame buffer indices this step draws into and reads back, as the base indexes them. + private const int NativeSsaoTargetIndex = 13; + private const int NativeSsaoBlurVerticalIndex = 14; + private const int NativeSsaoBlurHorizontalIndex = 15; + + /// + /// The AO step, natively. Reproduces + /// exactly: the platform's own + /// AO first, vanilla SSAO and its blur when that stood down, then the composite - under the + /// vanilla branch only while TAA is actually running, under the GTAO branch always. + /// + private void NativeAmbientOcclusion(float[] projectMatrix) + { + OptimumPostSsaoInScene = false; + OptimumPostAmbientOcclusionTexture = 0; + if (OptimumRenderSsao && projectMatrix != null) + { + OptimumPostAmbientOcclusionTexture = RenderOptimumAmbientOcclusion(projectMatrix); + } + + if (OptimumPostAmbientOcclusionTexture == 0 && OptimumRenderSsao && projectMatrix != null) + { + Size2i client = OptimumWindowClientSize(); + float ssaa = OptimumPostSsaaLevel; + + // Outside every native pass, exactly where the OpenGL body puts them: this is the + // GL-shaped state the steps after this one inherit. + GlToggleBlend(on: false); + NativeVanillaSsaoPass(projectMatrix, client, ssaa); + NativeBilateralBlurPasses(); + // The body's tail: the blur's last target is what it leaves bound, with the viewport + // back at full render resolution - the Luma step inherits that viewport. + LoadFrameBuffer(EnumFrameBuffer.SSAOBlurVertical); + GlToggleBlend(on: true); + GlViewport(0, 0, (int)(ssaa * client.Width), (int)(ssaa * client.Height)); + if (OptimumTaaRequested && TaaTargetsReady) + { + NativeSceneSsaoPass(); + } + } + + if (OptimumPostAmbientOcclusionTexture != 0) + { + NativeSceneSsaoPass(); + } + } + + /// + /// Whether the native route can run this step at all. The shader programs are the OpenGL + /// body's own - it uses them unguarded - so a frame that has none of them falls back to the + /// body rather than drawing nothing. + /// + private bool NativeAmbientOcclusionReady() + { + if (device == null) return false; + List buffers = FrameBuffers; + if (buffers == null || buffers.Count <= NativeSsaoBlurHorizontalIndex) return false; + if (buffers[0] == null || buffers[1] == null) return false; + if (!OptimumRenderSsao) return true; + + FrameBufferRef primary = buffers[0]; + if (primary.ColorTextureIds == null || primary.ColorTextureIds.Length < 4) return false; + if (buffers[NativeSsaoTargetIndex] == null || buffers[NativeSsaoBlurVerticalIndex] == null || + buffers[NativeSsaoBlurHorizontalIndex] == null) + { + return false; + } + + ShaderProgramSsao ssao = ShaderPrograms.Ssao; + ShaderProgramBilateralblur blur = ShaderPrograms.Bilateralblur; + if (ssao == null || ssao.LoadError || ssao.Disposed) return false; + if (blur == null || blur.LoadError || blur.Disposed) return false; + return true; + } + + // ------------------------------------------------------------------ 1: vanilla SSAO + + /// + /// The raw SSAO pass. One colour slot on frameBuffers[13], cleared white at pass entry the + /// way the body's ClearSsaoTarget clears it, no blend, no depth, and the viewport the body's + /// LoadFrameBuffer(SSAO) case sets - the target's own size. The four samplers and the four + /// uniform values are the body's, including the half-resolution screenSize fudge and the + /// temporal dither index, which is written under exactly the condition that compiles it in. + /// + private void NativeVanillaSsaoPass(float[] projectMatrix, Size2i client, float ssaa) + { + List buffers = FrameBuffers; + FrameBufferRef primary = buffers[0]; + FrameBufferRef transparent = buffers[1]; + FrameBufferRef target = buffers[NativeSsaoTargetIndex]; + + NativePipeline? pipeline = NativePostPipeline(nativeSsao, ShaderPrograms.Ssao, target.FboId, 1u, + NativeOpaqueSlotZeroBlend(), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) return; + + int gNormal = primary.ColorTextureIds[2]; + int gPosition = primary.ColorTextureIds[3]; + int noise = target.ColorTextureIds[1]; + int revealage = transparent.ColorTextureIds[1]; + + if (BeginNativeAoPass("SSAO/" + NativeSsaoTargetIndex, target.FboId, target.Width, target.Height, + new[] { gPosition, gNormal, noise, revealage }, clearWhite: true)) + { + // screenSize: the body's num is 0.5 at SSAA 1 and 1 otherwise, so the value is the + // SSAO target's resolution at SSAA 1 and the full render resolution above it. + float half = ssaa == 1f ? 0.5f : 1f; + device.WriteNative(pipeline, nativeSsao.Uniforms[0], ssaa * client.Width * half, ssaa * client.Height * half); + WriteNativeFloats(pipeline, nativeSsao.Uniforms[1], projectMatrix); + WriteNativeFloats(pipeline, nativeSsao.Uniforms[2], OptimumSsaoKernel); + if (OptimumConfig.EffectiveTaa) + { + device.WriteNative(pipeline, nativeSsao.Uniforms[3], + (float)(OptimumTemporal.Frame.FrameIndex & 1023L)); + } + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeSsao.Samplers[0], gPosition), + new NativeTexture(nativeSsao.Samplers[1], gNormal), + new NativeTexture(nativeSsao.Samplers[2], noise), + new NativeTexture(nativeSsao.Samplers[3], revealage), + }); + } + device.EndNativePass(); + } + + // ------------------------------------------------------------------ 2: bilateral blur + + /// + /// The bilateral blur ping-pong: one horizontal half-iteration into frameBuffers[15] and one + /// vertical into frameBuffers[14], once at SSAO quality 1 and three times otherwise. Each + /// half-iteration is its own pass, because each writes a target the next one samples. + /// + /// frameSize is the body's: frameBuffers[15]'s size, captured once and reused by every + /// half-iteration including the vertical ones that write frameBuffers[14]. The two targets are + /// built at the same size, so this is a value, not a bug to fix - and it is reproduced, not + /// corrected. depthTexture is likewise bound on both halves: the body sets it on the + /// horizontal half only and the vertical half inherits the same binding from the program. + /// + private void NativeBilateralBlurPasses() + { + List buffers = FrameBuffers; + FrameBufferRef primary = buffers[0]; + FrameBufferRef horizontal = buffers[NativeSsaoBlurHorizontalIndex]; + FrameBufferRef vertical = buffers[NativeSsaoBlurVerticalIndex]; + + NativePipeline? pipeline = NativePostPipeline(nativeBilateralBlur, ShaderPrograms.Bilateralblur, + horizontal.FboId, 1u, NativeOpaqueSlotZeroBlend(), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) return; + + int depth = primary.DepthTextureId; + int iterations = ClientSettings.SSAOQuality == 1 ? 1 : 3; + for (int i = 0; i < iterations; i++) + { + int source = buffers[i == 0 ? NativeSsaoTargetIndex : NativeSsaoBlurVerticalIndex].ColorTextureIds[0]; + NativeBilateralBlurHalf(pipeline, horizontal, source, depth, horizontal, isVertical: 0); + NativeBilateralBlurHalf(pipeline, vertical, horizontal.ColorTextureIds[0], depth, horizontal, isVertical: 1); + } + } + + /// One half-iteration of the blur: one target, one input, the shared frameSize. + private void NativeBilateralBlurHalf(NativePipeline pipeline, FrameBufferRef target, int source, int depth, + FrameBufferRef frameSizeSource, int isVertical) + { + if (!BeginNativeAoPass("SSAOBlur/" + target.FboId, target.FboId, target.Width, target.Height, + new[] { source, depth }, clearWhite: false)) + { + device.EndNativePass(); + return; + } + + device.WriteNative(pipeline, nativeBilateralBlur.Uniforms[0], + (float)frameSizeSource.Width, (float)frameSizeSource.Height); + device.WriteNative(pipeline, nativeBilateralBlur.Uniforms[1], isVertical); + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeBilateralBlur.Samplers[0], source), + new NativeTexture(nativeBilateralBlur.Samplers[1], depth), + }); + device.EndNativePass(); + } + + // ------------------------------------------------------------------ 3: the AO composite + + /// + /// The AO composite, natively: the visibility term multiplied into Primary colour 0 and + /// nothing else, before the TAA resolve reads that colour. + /// + /// The Multiply blend is the pipeline's - dst * (1 - srcAlpha), which is what + /// GlToggleBlend(true, EnumBlendMode.Multiply) sets - and the single written colour slot is + /// the pass's, not a draw-buffer mask. That is also what lets the pass sample Primary's + /// G-buffer position attachment: a slot the pass leaves out is not part of its scope, so it + /// is read as a texture rather than being feedback. + /// + private void NativeSceneSsaoPass() + { + ShaderProgram composite = ShaderPrograms.SceneSsao; + if (composite == null || composite.LoadError || composite.Disposed) return; + + List buffers = FrameBuffers; + FrameBufferRef primary = buffers[0]; + FrameBufferRef transparent = buffers[1]; + + int aoTexture = OptimumPostAmbientOcclusionTexture; + if (aoTexture == 0) + { + FrameBufferRef blurred = buffers[NativeSsaoBlurVerticalIndex]; + if (blurred?.ColorTextureIds == null || blurred.ColorTextureIds.Length == 0) return; + aoTexture = blurred.ColorTextureIds[0]; + } + + bool gtao = OptimumConfig.AmbientOcclusionShadersUseGtao; + int gPosition = gtao ? primary.ColorTextureIds[3] : 0; + int revealage = gtao ? transparent.ColorTextureIds[1] : 0; + + NativePipeline? pipeline = NativePostPipeline(nativeSceneSsao, composite, primary.FboId, 1u, + NativeMultiplySlotZeroBlend(), depthTest: false, depthWrite: false, CompareOp.Less); + if (pipeline == null) return; + + // The body binds Primary through LoadFrameBuffer here, which is also what puts the + // viewport back at full render resolution for the steps that follow. + LoadFrameBuffer(EnumFrameBuffer.Primary); + + var reads = new List { aoTexture }; + if (gtao) + { + reads.Add(gPosition); + reads.Add(revealage); + } + + if (BeginNativeAoPass("SceneSsao/" + primary.FboId, primary.FboId, primary.Width, primary.Height, + reads.ToArray(), clearWhite: false)) + { + var textures = new List { new(nativeSceneSsao.Samplers[0], aoTexture) }; + if (gtao) + { + textures.Add(new NativeTexture(nativeSceneSsao.Samplers[1], gPosition)); + textures.Add(new NativeTexture(nativeSceneSsao.Samplers[2], revealage)); + device.WriteNative(pipeline, nativeSceneSsao.Uniforms[1], + OptimumPostAmbientOcclusionTexture != 0 ? 1 : 0); + } + device.WriteNative(pipeline, nativeSceneSsao.Uniforms[0], 1f / primary.Height); + device.DrawNativeFullscreen(pipeline, textures.ToArray()); + } + device.EndNativePass(); + + // The net GL-shaped state the body's composite leaves: blending back on in the standard + // mode and the depth test back on. The draw-buffer mask is not restored because it was + // never narrowed - the written slot is the pass's, so the world mask never moved. + GlToggleBlend(on: true); + GlEnableDepthTest(); + OptimumPostSsaoInScene = true; + } + + // ------------------------------------------------------------------ shared plumbing + + /// Colour slot 0 alone, opaque: no blend, every channel written, every other slot masked out. + private static AttachmentBlend[] NativeOpaqueSlotZeroBlend() => new[] { AttachmentBlend.Default }; + + /// + /// Colour slot 0 alone with EnumBlendMode.Multiply: glBlendFuncSeparate(ZERO, + /// ONE_MINUS_SRC_ALPHA, ONE, ONE_MINUS_SRC_ALPHA) over glBlendEquation(FUNC_ADD). + /// + private static AttachmentBlend[] NativeMultiplySlotZeroBlend() + { + AttachmentBlend blend = AttachmentBlend.Default; + blend.Enabled = true; + blend.SrcColor = BlendFactor.Zero; + blend.DstColor = BlendFactor.OneMinusSrcAlpha; + blend.ColorOp = BlendOp.Add; + blend.SrcAlpha = BlendFactor.One; + blend.DstAlpha = BlendFactor.OneMinusSrcAlpha; + blend.AlphaOp = BlendOp.Add; + return new[] { blend }; + } + + /// + /// A pass of this step: one written colour slot, its own viewport (every target here is drawn + /// at its own size, which is what the body's LoadFrameBuffer cases set), its reads, and the + /// white clear the SSAO target starts from. + /// + private bool BeginNativeAoPass(string name, int framebufferId, int width, int height, int[] reads, bool clearWhite) => + device.BeginNativePass(new NativePassDescription + { + Name = name, + FramebufferId = framebufferId, + ColorSlots = 1u, + Reads = reads, + Flags = PassFlags.None, + ClearSlots = clearWhite ? 1u : 0u, + ClearValue = new[] { 1f, 1f, 1f, 1f }, + ViewportWidth = width, + ViewportHeight = height, + }); + +} + +// Optimum AO (docs/vulkan.md#ambient-occlusion section C): the GTAO visibility-bitmask pass +// on the device. The base's RenderPostprocessingEffects asks for it where vanilla SSAO would run +// and composes the returned texture through ApplyOptimumSceneSsao before the TAA resolve. +public partial class VulkanClientPlatform +{ + private GtaoRenderer? ambientOcclusion; + + /// This frame's output, or 0 when GTAO did not run (the debug outputs and the compose pass's reads key off it). + private int ambientOcclusionOutput; + + /// The output texture whose sampler state was last set up. + private int ambientOcclusionSampledTexture; + + /// Set when the passes cannot run on this device; vanilla SSAO then runs for the session. + private string? ambientOcclusionFailure; + + private bool ambientOcclusionToneRefusalLogged; + + private (string Preset, bool Temporal, GtaoSettings Settings)? ambientOcclusionSettingsCache; + + /// + /// Runs GTAO when the live shaders were built for it (OPTIMUMAO, stamped from + /// and the SSAO G-buffer's condition) and returns + /// the denoised visibility; 0 hands the frame to vanilla SSAO. + /// + public override int RenderOptimumAmbientOcclusion(float[] projectMatrix) + { + ambientOcclusionOutput = 0; + if (device == null || projectMatrix == null || ambientOcclusionFailure != null) return 0; + if (!OptimumConfig.AmbientOcclusionShadersUseGtao) return 0; + FrameBufferRef? primary = FrameBuffers is { Count: > 0 } buffers ? buffers[0] : null; + if (primary?.ColorTextureIds == null || primary.ColorTextureIds.Length < 4 || primary.DepthTextureId == 0) return 0; + + // Avoid feeding high-variance rotating AO into the scene TAA. Keep + // the AO sampling phase fixed and denoise before scene composition. + bool temporal = OptimumConfig.EffectiveTaa && TaaTargetsReady; + GtaoSettings settings = AmbientOcclusionSettings(temporal); + if (!ambientOcclusionToneRefusalLogged && settings.EffectiveTone(0, out string? refusal) != settings.Tone) + { + ambientOcclusionToneRefusalLogged = true; + Logger.Warning("[Optimum] AO: " + refusal); + } + const uint noiseIndex = 0u; + + ambientOcclusion ??= new GtaoRenderer(device); + int output = ambientOcclusion.Render(primary.DepthTextureId, primary.ColorTextureIds[2], projectMatrix, settings, noiseIndex); + if (output == 0) + { + if (ambientOcclusion.ProgramsFailed) + { + ambientOcclusionFailure = ambientOcclusion.LastError ?? "unknown"; + Logger.Error("[Optimum] AO: GTAO is unavailable on this device, vanilla SSAO runs instead: " + ambientOcclusionFailure); + } + return 0; + } + if (output != ambientOcclusionSampledTexture) + { + // Composed with texelFetch at the same resolution; nearest and clamp keep any sampling exact. + SetupOptimumTextureSampler(output, 9728, 33071); + ambientOcclusionSampledTexture = output; + } + ambientOcclusionOutput = output; + return output; + } + + /// The debug outputs of this frame in the base's index order (working term, edges, depth level 0, output). + public override int OptimumAmbientOcclusionDebugTexture(int index) + { + if (ambientOcclusionOutput == 0 || ambientOcclusion == null) return 0; + return index switch + { + 0 => ambientOcclusion.WorkingTermTexture, + 1 => ambientOcclusion.EdgesTexture, + 2 => ambientOcclusion.WorkingDepthTexture, + 3 => ambientOcclusion.OutputTexture, + _ => 0, + }; + } + + /// The preset with the measurement overrides from the environment, rebuilt only when the preset or TAA changes. + private GtaoSettings AmbientOcclusionSettings(bool temporal) + { + string preset = OptimumConfig.AmbientOcclusionPreset ?? ""; + if (ambientOcclusionSettingsCache is { } cached && cached.Preset == preset && cached.Temporal == temporal) + { + return cached.Settings; + } + GtaoPreset parsed = GtaoSettings.ParsePreset(preset); + GtaoSettings settings = (temporal + ? GtaoSettings.ForStableTemporal(parsed) + : GtaoSettings.ForPreset(parsed, temporal: false)) + .WithEnvironment(Environment.GetEnvironmentVariable); + ambientOcclusionSettingsCache = (preset, temporal, settings); + return settings; + } + + /// The size-dependent targets go with the framebuffers; the next frame recreates them. + private void ReleaseAmbientOcclusionTargets() + { + ambientOcclusion?.ReleaseTargets(); + ambientOcclusionOutput = 0; + ambientOcclusionSampledTexture = 0; + } + + private void ReleaseAmbientOcclusion() + { + ambientOcclusion?.Dispose(); + ambientOcclusion = null; + ambientOcclusionOutput = 0; + ambientOcclusionSampledTexture = 0; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeWorld.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeWorld.cs new file mode 100644 index 00000000..3e757002 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.NativeWorld.cs @@ -0,0 +1,856 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native render systems (docs/vulkan.md), Phase 3b decision 5 +// stage 2: the sky systems that are not the dome, the particle pools and the decal pool, all +// on the native device API the sky dome proved. +// +// What they draw, and where the other side is: +// - the night sky box - SystemRenderNightSky's 75-unit star cube, seam +// ClientPlatformAbstract.RenderNightSkyBox, neutral body RenderMesh; +// - the moon - SystemRenderSunMoon's quad under celestialobject, seam +// ClientPlatformAbstract.RenderCelestialQuad, neutral body RenderMesh; +// - the cube particles - SystemRenderParticles' instanced pool draw on Primary, seam +// ClientPlatformAbstract.RenderParticles, neutral body +// RenderMeshInstanced; +// - the decals - SystemRenderDecals' pooled multi-draw, scope seam +// ClientPlatformAbstract.BeginDecalPass / EndDecalPass with empty +// neutral bodies: the lib runs the vanilla MeshDataPool.Draw between +// them and the pool's RenderMesh multi-draw is taken natively while +// the scope is open (the mesh handle is internal in the vanilla API, +// so it can never be a seam parameter). +// NativeWorldEnabled false takes the neutral body on the Vulkan device too, which is the route +// the differential tests compare against. +// +// Target and slots: every one of them draws into the framebuffer its stage has bound - Primary +// for all four. The colour slots are NativeWorldPassColorSlots: the same set the emulated route's +// draw-buffer mask would hold, derived from the platform's own motion-window state +// (MotionAttachmentIndex, OptimumMotionWriteActive) rather than from the GL state tracker +// (decision 3). That is what makes a cube particle's and a decal's motion vector land through +// the one writer include exactly while their caller's window is open, and keeps the motion +// attachment out of the night sky's and the moon's scope entirely. +// +// State that is not obvious, per system, and where it comes from: +// - the night sky and the moon run with the depth test off, because their callers call +// GlDisableDepthTest; GL writes no depth with the test off, so depth writes are off too. +// The night sky also disables culling itself. The moon inherits the cull state the night +// sky left, which is off - no vanilla Opaque renderer between them turns it back on. +// - the cube particles and the decals run with the depth test and depth writes on: the Opaque +// stage is entered with ChunkRenderer.RenderOpaque's depth mask and test, and the AfterOIT +// stage is entered with ClientMain's own GlDepthMask/GlEnableDepthTest. Culling is off for +// both: SystemRenderNightSky leaves it off for the rest of the Opaque stage, and +// SystemRenderDecals calls GlDisableCullFace itself. +// - blending is on in the standard mode for the moon, the cube particles and the decals +// (their callers call GlToggleBlend(on: true)), and off for the night sky. +// - the motion attachment never blends. Inside a motion window the native pass states +// replace-blending on that one attachment per attachment, which is what +// ApplyOptimumMotionBlendState does for an emulated draw. +// +// What pins them: NativeWorldSystemsTests (old route against native route, including the motion +// attachment bit for bit) and Optimum.Tests/native-world-systems-coverage-tests.cs. +public partial class VulkanClientPlatform +{ + /// + /// False runs each seam's neutral body - the OpenGL body's own draw - on the Vulkan device + /// instead of the native pass: the old route the differential tests compare against, in the + /// pattern of . + /// + internal bool NativeWorldEnabled { get; set; } = Environment.GetEnvironmentVariable("OPTIMUM_VK_NATIVE_WORLD") != "0"; + + /// The star cube's pipeline: no per-draw uniform, one samplerCube. + private readonly NativeMeshPass nativeNightSky = + new("nightsky", Array.Empty(), new[] { "ctex" }); + + /// + /// The moon's pipeline. "tex" is the body's own texture; "sky" and "glow" are the frame + /// textures skycolor.fsh reads to shade the body against the sky behind it, and they are + /// passed here because a native draw resolves what it samples from handles and nothing else + /// in this pass would refresh the frame table's entries for them. + /// + private readonly NativeMeshPass nativeCelestial = + new("celestialobject", Array.Empty(), new[] { "tex", "sky", "glow" }); + + /// + /// The sun's pipeline, through the standard program. Its samplers are resolved from the + /// pipeline's own declaration (), because standard also reads frame + /// textures a native draw has to name by handle. + /// + private readonly NativeMeshPass nativeSun = + new("standard", Array.Empty(), Array.Empty()); + + /// + /// The quad particle pool's pipeline: drawn in the OIT stage onto the Transparent target, under + /// the blend contract the client applied to that target (the chunk route records it), with + /// depth test on and depth writes off - what LoadFrameBuffer(Transparent) sets. + /// + private readonly NativeMeshPass nativeParticlesQuad = + new("particlesquad", Array.Empty(), Array.Empty()); + + /// Any plain RenderMesh under the vanilla standard program (RenderStandardMeshNative). + private readonly NativeMeshPass nativeStandardGui = + new("standard", Array.Empty(), Array.Empty()); + + private readonly NativeMeshPass nativeStandardMesh = + new("standard", Array.Empty(), Array.Empty()); + + /// The cube particle pool's pipeline: no per-draw uniform and no sampler at all. + private readonly NativeMeshPass nativeParticlesCube = + new("particlescube", Array.Empty(), Array.Empty()); + + /// + /// The decal pool's pipeline. origin and modelViewMatrix are DRAW uniforms the client system + /// already set through the program's own setters, so they ride in the program's push shadow + /// and the pass writes nothing per draw; the two atlases are its samplers. + /// + private readonly NativeMeshPass nativeDecals = + new("decals", Array.Empty(), new[] { "decalTexture", "blockTexture" }); + + // ------------------------------------------------------------------ shared derivations + + /// + /// The colour slots a world pass writes: the set the emulated route's draw-buffer mask would + /// hold at this point in the frame, computed from the platform's own motion-window state + /// rather than read back out of the tracker. + /// + /// Outside Primary, and with TAA off ( + /// negative), that is every bound colour slot. On Primary with TAA on it is Primary's default + /// colour set, plus the motion attachment exactly while a motion window is open - the two + /// sets and + /// switch between. + /// + private uint NativeWorldPassColorSlots(FrameBufferRef target) + { + if (IsTransparentTarget(target) && nativeTransparentSlots != 0) return nativeTransparentSlots; + uint all = NativeAllColorSlots(target); + int motion = MotionAttachmentIndex; + if (motion < 0 || motion >= 32) return all; + + List buffers = FrameBuffers; + if (buffers == null || buffers.Count == 0 || !ReferenceEquals(target, buffers[0])) return all; + + uint mask = OptimumMotionWriteActive + ? (1u << (motion + 1)) - 1u + : (1u << motion) - 1u; + return mask & all; + } + + /// + /// A world pass's per-attachment blend: the caller's blend mode on every colour attachment, + /// except the motion attachment inside an open motion window, which replaces rather than + /// blends. A blended motion vector is a weighted average of two surfaces' displacements and + /// belongs to neither, which is why forces + /// (ONE, ZERO) with FUNC_ADD there on the emulated route; this states the same thing on the + /// pipeline instead of through a tracked toggle. + /// + private AttachmentBlend[] NativeWorldBlend(RenderTargetFormats formats, bool blending) + { + var blend = new AttachmentBlend[Math.Max(formats.ColorFormats.Length, 1)]; + for (int i = 0; i < blend.Length; i++) + { + blend[i] = AttachmentBlend.Default; + blend[i].Enabled = blending; + } + + int motion = MotionAttachmentIndex; + if (OptimumMotionWriteActive && motion >= 0 && motion < blend.Length) + { + blend[motion] = AttachmentBlend.Default; + blend[motion].Enabled = blending; + blend[motion].SrcColor = BlendFactor.One; + blend[motion].DstColor = BlendFactor.Zero; + blend[motion].SrcAlpha = BlendFactor.One; + blend[motion].DstAlpha = BlendFactor.Zero; + } + return blend; + } + + /// + /// Everything a native world draw needs before it can be recorded: the target the stage + /// bound, the program it is drawing with, the mesh and its vertex layout, the colour slots + /// and their formats, and the pipeline for that combination. False means the caller takes + /// its seam's neutral body, which is always a legal answer. + /// + private bool NativeWorldPrepare(NativeMeshPass pass, MeshRef mesh, bool blending, bool depth, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline, + Func? blendFor = null, bool? depthWrite = null, + CompareOp? depthCompare = null, CullModeFlags? cull = null, bool samplesBoundDepth = false) + { + target = null!; + vao = null!; + slots = 0; + pipeline = null!; + + if (!NativeWorldEnabled || device == null || mesh == null) return false; + + FrameBufferRef bound = CurrentFrameBuffer; + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + var buffers = mesh as VAO; + if (bound == null || program == null || buffers == null || buffers.VaoId == 0 || buffers.Disposed) + { + return false; + } + + int layoutId = device.NativeMeshLayoutId(buffers.VaoId); + if (layoutId < 0) return false; + + uint colorSlots = NativeWorldPassColorSlots(bound); + if (colorSlots == 0) return false; + + RenderTargetFormats? formats = device.NativeTargetFormats(bound.FboId, colorSlots); + if (formats == null) return false; + + NativePipeline? built = NativeMeshPipelineFor(pass, program, bound.FboId, colorSlots, layoutId, + new NativePipelineDescription + { + Blend = blendFor != null ? blendFor(formats.ColorFormats.Length) : NativeWorldBlend(formats, blending), + DepthTest = depth, + DepthWrite = depthWrite ?? depth, + DepthCompare = depthCompare ?? CompareOp.Less, + Cull = cull ?? CullModeFlags.None, + Topology = PrimitiveTopology.TriangleList, + SamplesBoundDepth = samplesBoundDepth, + }); + if (built == null) return false; + + target = bound; + vao = buffers; + slots = colorSlots; + pipeline = built; + return true; + } + + /// + /// Opens the native pass for one world draw on the target its stage bound, with the reads it + /// samples declared outright. + /// + private bool NativeWorldBeginPass(string name, FrameBufferRef target, uint slots, int[] reads) + { + Rect2D viewport = StatedViewport(); + return device.BeginNativePass(new NativePassDescription + { + Name = name + "/" + target.FboId, + FramebufferId = target.FboId, + ColorSlots = slots, + Reads = reads, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + }); + } + + /// + /// Closes the native pass and declares the stage's own pass context again, because every + /// renderer after this one draws into the same target through the generic stated route - the same + /// restoration the sky dome's pass and the TAA resolve's do. + /// + private void NativeWorldEndPass(FrameBufferRef target, string outer, PassFlags outerFlags) + { + device.EndNativePass(); + SetPassContext(outer, outerFlags); + } + + // ------------------------------------------------------------------------ the seams + + /// The star cube's draw: the native pass, or the seam's neutral body. + public override void RenderNightSkyBox(MeshRef nightSkyBox, int cubeTextureId) + { + // No depth and no blending: SystemRenderNightSky has called GlDisableDepthTest and + // GlDisableCullFace, and nightsky.fsh writes opaque colour into slot 0. + if (!NativeWorldPrepare(nativeNightSky, nightSkyBox, blending: false, depth: false, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline)) + { + base.RenderNightSkyBox(nightSkyBox, cubeTextureId); + return; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + if (NativeWorldBeginPass("NightSky", target, slots, new[] { cubeTextureId })) + { + // The cube map resolves into the bindless table's cube array, which the sampler's + // own kind selects - a 2D texture bound here would be refused rather than sampled. + device.DrawNativeMesh(pipeline, vao.VaoId, new[] + { + new NativeTexture(nativeNightSky.Samplers[0], cubeTextureId), + }); + } + NativeWorldEndPass(target, outer, outerFlags); + } + + /// The moon's draw: the native pass, or the seam's neutral body. + public override void RenderCelestialQuad(MeshRef quad, int bodyTextureId, int skyTextureId, int glowTextureId) + { + // Blending on in the standard mode and no depth: SystemRenderSunMoon has called + // GlToggleBlend(on: true), GlDisableCullFace and GlDisableDepthTest before both bodies. + if (!NativeWorldPrepare(nativeCelestial, quad, blending: true, depth: false, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline)) + { + base.RenderCelestialQuad(quad, bodyTextureId, skyTextureId, glowTextureId); + return; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + if (NativeWorldBeginPass("Celestial", target, slots, new[] { bodyTextureId, skyTextureId, glowTextureId })) + { + device.DrawNativeMesh(pipeline, vao.VaoId, new[] + { + new NativeTexture(nativeCelestial.Samplers[0], bodyTextureId), + new NativeTexture(nativeCelestial.Samplers[1], skyTextureId), + new NativeTexture(nativeCelestial.Samplers[2], glowTextureId), + }); + } + NativeWorldEndPass(target, outer, outerFlags); + } + + /// + /// The sun's visible quad: the native pass, or the seam's neutral body. + /// What it draws: the sun disc of SystemRenderSunMoon.OnRenderFrame3D. The other side: + /// ClientPlatformAbstract.RenderSunQuad, whose neutral body is the RenderMesh it replaced. + /// Target and slots: the stage's bound target and . + /// State: blended in the standard mode, no depth test, no culling - SystemRenderSunMoon's + /// GlToggleBlend(on: true), GlDisableDepthTest and GlDisableCullFace, stated on the pipeline. + /// Only the registered vanilla standard program is taken: a mod can register its own program + /// under the same pass name (VSEssentials' first-person item shader does). + /// What pins it: NativeWorldSystemsTests.TheSunMatchesTheSeamsNeutralBody. + /// + public override void RenderSunQuad(MeshRef quad, int sunTextureId) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (program == null || !ReferenceEquals(program, ShaderPrograms.Standard) || + !NativeWorldPrepare(nativeSun, quad, blending: true, depth: false, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline)) + { + base.RenderSunQuad(quad, sunTextureId); + return; + } + + string[] names = pipeline.SamplerNames; + var textures = new NativeTexture[names.Length]; + var reads = new int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int id = names[i] == "tex" ? sunTextureId : DeclaredProgramTexture(program.ProgramId, names[i]); + textures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + reads[i] = id; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + if (NativeWorldBeginPass("Sun", target, slots, reads)) + { + device.DrawNativeMesh(pipeline, vao.VaoId, textures); + } + NativeWorldEndPass(target, outer, outerFlags); + } + + /// + /// One particle pool's instanced draw: the native pass, or the seam's neutral body. + /// + /// Both pools take the native route. The cube pool draws on Primary here; the quad pool draws + /// in the OIT stage onto the Transparent target and goes through , + /// which states the Transparent blend contract the client applied. A draw under any other + /// program falls through to the neutral body: the pipeline request names the vanilla program. + /// + public override void RenderParticles(MeshRef model, int quantity, int particleTextureId) + { + if (quantity <= 0) + { + base.RenderParticles(model, quantity, particleTextureId); + return; + } + + if (ReferenceEquals(ShaderProgramBase.CurrentShaderProgram, ShaderPrograms.Particlesquad)) + { + RenderQuadParticles(model, quantity, particleTextureId); + return; + } + + // Blending on in the standard mode (the caller's GlToggleBlend) and the Opaque stage's + // depth test and depth writes, which ChunkRenderer.RenderOpaque established and no + // renderer between it and the particles turns off again. + if (!NativeWorldPrepare(nativeParticlesCube, model, blending: true, depth: true, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline)) + { + base.RenderParticles(model, quantity, particleTextureId); + return; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + // Inside the caller's motion window the motion attachment is one of the pass's colour + // slots (NativeWorldPassColorSlots) and replaces rather than blends (NativeWorldBlend), so + // particlescube.fsh's motion.glsl writer lands exactly what it lands on the GL path. + if (NativeWorldBeginPass("Particles", target, slots, Array.Empty())) + { + device.DrawNativeMeshInstanced(pipeline, vao.VaoId, quantity, ReadOnlySpan.Empty); + } + NativeWorldEndPass(target, outer, outerFlags); + } + + /// + /// The quad particle pool's instanced draw in the OIT stage: the native pass, or the seam's + /// neutral body. The other side is ClientPlatformAbstract.RenderParticles' neutral body, drawn + /// under the state LoadFrameBuffer(Transparent) and ApplyTransparentPassBlendState left. + /// Target and slots: the Transparent target, every bound slot; the pipeline masks the outputs + /// particlesquad does not write. State: the Transparent blend contract the client last applied + /// (weighted accumulation, revealage, glow), depth test on, depth writes off, no culling. + /// Falls back while that contract has not been recorded or the bound target is not + /// Transparent. particleTex resolves from the program's declared texture: the lib binds it + /// through the program's setter and hands the seam 0. + /// + private void RenderQuadParticles(MeshRef model, int quantity, int particleTextureId) + { + FrameBufferRef bound = CurrentFrameBuffer; + AttachmentBlend[]? contract = nativeTransparentBlend; + if (bound == null || contract == null || !IsTransparentTarget(bound) || + !NativeWorldPrepare(nativeParticlesQuad, model, blending: true, depth: true, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline, + count => + { + var blend = new AttachmentBlend[Math.Max(count, 1)]; + for (int i = 0; i < blend.Length; i++) + { + blend[i] = i < contract.Length ? contract[i] : AttachmentBlend.Default; + blend[i].Enabled = true; + } + return blend; + }, + depthWrite: false)) + { + base.RenderParticles(model, quantity, particleTextureId); + return; + } + + ShaderProgramBase program = ShaderProgramBase.CurrentShaderProgram!; + string[] names = pipeline.SamplerNames; + var textures = new NativeTexture[names.Length]; + var reads = new int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int id = names[i] == "particleTex" && particleTextureId != 0 + ? particleTextureId + : DeclaredProgramTexture(program.ProgramId, names[i]); + textures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + reads[i] = id; + } + + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + if (NativeWorldBeginPass("ParticlesOit", target, slots, reads)) + { + device.DrawNativeMeshInstanced(pipeline, vao.VaoId, quantity, textures); + } + NativeWorldEndPass(target, outer, outerFlags); + } + + /// + /// The per-attachment blend a world draw inherits from what the client stated: the stated + /// mode on every slot, with GlToggleBlend's own exceptions - the SSAO G-buffer slots and the + /// open motion attachment replace rather than blend - and on the Transparent target the + /// recorded OIT contract with the stated enable. + /// + private AttachmentBlend[] StatedWorldBlend(FrameBufferRef target, int count) + { + var blend = new AttachmentBlend[Math.Max(count, 1)]; + AttachmentBlend[]? contract = nativeTransparentBlend; + bool transparent = contract != null && IsTransparentTarget(target); + bool primary = IsPrimaryTarget(target); + int motion = primary && OptimumMotionWriteActive ? MotionAttachmentIndex : -1; + for (int i = 0; i < blend.Length; i++) + { + if (transparent) + { + blend[i] = i < contract!.Length ? contract[i] : AttachmentBlend.Default; + blend[i].Enabled = statedBlendOn; + } + else if (primary && statedBlendOn && OptimumRenderSsao && (i == 2 || i == 3)) + { + blend[i] = ReplaceBlend(true); + } + else + { + blend[i] = AttachmentBlend.For(statedBlendOn, statedBlendMode); + // World/UI separation: a world-program draw into the UI image (held items in + // a dialog, the reticle's disc) accumulates coverage like the GUI does. + if (stated.IsUiImage(target?.FboId ?? PassDeclaration.DefaultFramebuffer)) + { + blend[i] = blend[i].ForUiImage(); + } + } + if (i == motion) blend[i] = ReplaceBlend(statedBlendOn); + blend[i].WriteMask &= ~statedColorMaskOff; + } + return blend; + } + + /// + /// A plain RenderMesh under the vanilla standard program - held and dropped items, block + /// entity models, the sun's disc outside its seam - recorded natively under the state the + /// client stated: blend, depth test, depth mask, depth function, cull. Every sampler the + /// pipeline declares resolves from the program's declared textures. False: the caller runs the + /// emulated draw. The colour mask the client stated is applied per slot. An open occlusion + /// query (the sun probe) is carried across the native pass: the pass opens its scope through + /// the target manager, whose scope hooks suspend the query in the closing scope and resume it + /// in the native one. The default framebuffer (GUI item icons) goes to + /// . + /// + private bool TryRenderStandardMeshNative(MeshRef mesh) + { + ShaderProgramBase? program = ShaderProgramBase.CurrentShaderProgram; + if (!NativeWorldEnabled || device == null || mesh == null || program == null || + !ReferenceEquals(program, ShaderPrograms.Standard)) + { + return false; + } + + FrameBufferRef bound = CurrentFrameBuffer; + CullModeFlags cull = statedCull + ? (statedCullBack ? CullModeFlags.BackBit : CullModeFlags.FrontBit) + : CullModeFlags.None; + if (bound == null) + { + return TryRenderStandardMeshToDefault(program, mesh, cull); + } + if (!NativeWorldPrepare(nativeStandardMesh, mesh, blending: statedBlendOn, depth: statedDepthTest, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline, + count => StatedWorldBlend(bound, count), depthWrite: statedDepthWrite, + depthCompare: GlEnums.CompareOpFrom(statedDepthFunc), cull: cull)) + { + return false; + } + + string[] names = pipeline.SamplerNames; + var textures = new NativeTexture[names.Length]; + var reads = new int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int id = DeclaredProgramTexture(program.ProgramId, names[i]); + textures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + reads[i] = id; + } + + string outer = passContext; + PassFlags outerFlags = passContextFlags; + bool drawn = false; + if (NativeWorldBeginPass("Standard", target, slots, reads)) + { + drawn = device.DrawNativeMesh(pipeline, vao.VaoId, textures); + } + NativeWorldEndPass(target, outer, outerFlags); + return drawn; + } + + /// + /// A standard-program draw into the default framebuffer: the GUI's item icons (hotbar, + /// inventory, held-item slots), which InventoryItemRenderer draws in the Ortho stage with + /// CurrentFrameBuffer null. The OpenGL side is ClientPlatformWindows.RenderMesh. Slot 0 takes + /// the stated blend through the tracker's factor table; depth test, mask, function, cull and + /// scissor are what the client stated (item icons use depth to sort their own faces). + /// + private AttachmentBlend[] StatedGuiSlots(RenderTargetFormats formats) + { + AttachmentBlend[] slots = GuiSlots(formats, statedBlendOn, PassDeclaration.DefaultFramebuffer, statedBlendMode); + slots[0].WriteMask &= ~statedColorMaskOff; + return slots; + } + + private bool TryRenderStandardMeshToDefault(ShaderProgramBase program, MeshRef mesh, CullModeFlags cull) + { + var vao = mesh as VAO; + if (vao == null || vao.VaoId == 0 || vao.Disposed) return false; + int framebufferId = PassDeclaration.DefaultFramebuffer; + int layoutId = device.NativeMeshLayoutId(vao.VaoId); + if (layoutId < 0) return false; + RenderTargetFormats? formats = device.NativeTargetFormats(framebufferId, 1u); + if (formats == null) return false; + + NativePipeline? pipeline = NativeMeshPipelineFor(nativeStandardGui, program, framebufferId, 1u, layoutId, + new NativePipelineDescription + { + Blend = StatedGuiSlots(formats), + DepthTest = statedDepthTest, + DepthWrite = statedDepthWrite, + DepthCompare = GlEnums.CompareOpFrom(statedDepthFunc), + Cull = cull, + Topology = device.NativeMeshTopology(vao.VaoId), + }); + if (pipeline == null) return false; + + string[] names = pipeline.SamplerNames; + var textures = new NativeTexture[names.Length]; + var reads = new int[names.Length]; + for (int i = 0; i < names.Length; i++) + { + int id = DeclaredProgramTexture(program.ProgramId, names[i]); + textures[i] = new NativeTexture(pipeline.Sampler(names[i]), id); + reads[i] = id; + } + + string outer = passContext; + PassFlags outerFlags = passContextFlags; + Rect2D viewport = StatedViewport(); + bool drawn = false; + if (device.BeginNativePass(new NativePassDescription + { + Name = "StandardGui/" + framebufferId, + FramebufferId = framebufferId, + ColorSlots = 1u, + Reads = reads, + Flags = PassFlags.AllowSplit, + ViewportX = viewport.Offset.X, + ViewportY = viewport.Offset.Y, + ViewportWidth = (int)viewport.Extent.Width, + ViewportHeight = (int)viewport.Extent.Height, + Scissor = scissorEnabled ? statedScissor : null, + })) + { + drawn = device.DrawNativeMesh(pipeline, vao.VaoId, textures); + } + device.EndNativePass(); + SetPassContext(outer, outerFlags); + return drawn; + } + + // ------------------------------------------------------------------- the decal scope + + /// True between and . + private bool decalScopeActive; + + /// The decal atlas handle the open scope passed in. + private int decalScopeDecalTextureId; + + /// The block atlas handle the open scope passed in. + private int decalScopeBlockTextureId; + + /// + /// Opens the decal pool's scope: the two atlas handles, which a native pass resolves what it + /// samples from, instead of from the units ShaderProgramDecals' setters bound them to. + /// + /// The mesh handle is deliberately not a parameter. MeshDataPool.modelRef is internal in the + /// vanilla API and the shipped VintagestoryAPI-patched.dll is vanilla plus api-patcher.cs's + /// hooks only, so a new public member on MeshDataPool would never reach the running client + /// (it did not, and the shipped client threw MissingMethodException on both backends). The + /// lib therefore runs the vanilla MeshDataPool.Draw, whose own + /// capi.Render.RenderMesh(modelRef, starts, sizes, count) lands in this platform's + /// override, and that override + /// routes to while this scope is open. The caller's + /// Draw runs the pool's cull first on both routes, so both draw the same ranges and only the + /// draw command differs. + /// + /// Where the other side is: ClientPlatformAbstract.BeginDecalPass / EndDecalPass have empty + /// neutral bodies, so the OpenGL path is vanilla MeshDataPool.Draw into + /// ClientPlatformWindows.RenderMesh -> GL.MultiDrawElements, exactly as before the seam. + /// Target and slots: Primary, inside SystemRenderDecals' motion window - a decal nudges the + /// depth buffer in front of the block it sits on and writes that surface's motion vector + /// itself, so the motion attachment is one of NativeWorldPassColorSlots and replaces rather + /// than blends (NativeWorldBlend). + /// State that is not obvious: standard blending on and the AfterOIT stage's depth test and + /// depth writes, which ClientMain sets before the stage; SystemRenderDecals turns culling + /// off itself. + /// What pins it: NativeWorldSystemsTests (old route against native route, including the + /// motion attachment bit for bit) and Optimum.Tests/native-world-systems-coverage-tests.cs. + /// + public override void BeginDecalPass(int decalTextureId, int blockTextureId) + { + decalScopeActive = true; + decalScopeDecalTextureId = decalTextureId; + decalScopeBlockTextureId = blockTextureId; + } + + /// Closes the scope opened. + public override void EndDecalPass() + { + decalScopeActive = false; + decalScopeDecalTextureId = 0; + decalScopeBlockTextureId = 0; + } + + /// + /// The decal pool's multi-draw, recorded natively, when it arrives through + /// inside an open decal scope. + /// False means the scope is closed, the native route is off, or the pass could not be + /// prepared, and the caller takes the generic stated multi-draw. + /// + internal bool TryDrawDecalPoolNative(MeshRef decalMesh, int[] indicesStarts, int[] indicesSizes, int groupCount) + { + if (!decalScopeActive) return false; + if (groupCount <= 0 || indicesStarts == null || indicesSizes == null) return false; + + if (!NativeWorldPrepare(nativeDecals, decalMesh, blending: true, depth: true, + out FrameBufferRef target, out VAO vao, out uint slots, out NativePipeline pipeline)) + { + return false; + } + + string outer = passContext; + PassFlags outerFlags = passContextFlags; + if (NativeWorldBeginPass("Decals", target, slots, + new[] { decalScopeDecalTextureId, decalScopeBlockTextureId })) + { + device.DrawNativeMeshMulti(pipeline, vao.VaoId, indicesStarts, indicesSizes, groupCount, new[] + { + new NativeTexture(nativeDecals.Samplers[0], decalScopeDecalTextureId), + new NativeTexture(nativeDecals.Samplers[1], decalScopeBlockTextureId), + }); + } + NativeWorldEndPass(target, outer, outerFlags); + return true; + } +} + +// The generic native draw: any program, drawn with the state the client stated through this +// platform's virtuals (StatedRenderState) - the last route of every draw virtual. The dedicated +// routes (chunks, entities, sky, particles, GUI, clouds, the post chain) keep their own contracts +// and run first; this one takes everything they do not recognise: mod renderers with their own +// programs, the vanilla programs without a dedicated route (aurora, block highlights, held item, +// lines, wireframe, the debug views) and the seams' neutral bodies behind the route switches. +// +// What it states, and from where: StatedDraw (the target the client addressed, its attached colour +// slots, the stated fixed-function state and texture units). All of it is client statements, none +// read back from the device. +// The clears (ClearTargetColor, ClearTargetDepth) are stated the same way: an explicit target, and +// OpenGL's rules - a clear writes only a selected draw buffer, not through an all-false colour +// mask, and a depth clear honours the depth mask. No state reaches the device any more; a draw +// this route cannot record is dropped and reported once. +// Pinned by Optimum.Tests/native-world-systems-coverage-tests.cs. +public partial class VulkanClientPlatform +{ + /// The fixed-function state the client stated, with OpenGL's semantics. + internal readonly StatedRenderState stated = new(); + + private readonly HashSet statedRefusalReported = new(); + + /// The program the client last used (glUseProgram); 0: none. + internal int statedProgram; + + /// The pass a running mod pass declared: the generic draws inside it are recorded under it. + internal PassDeclaration? statedPass; + + /// Draws the generic route recorded, and draws it could not record. Tests read them. + internal long StatedDrawsForTests { get; private set; } + + + // ------------------------------------------------------------------ recording helpers + + /// The draw buffers the client selects for a framebuffer (0 is the default target). + internal void StateDrawBuffers(int framebufferId, int mask) + { + stated.SetDrawBuffers(framebufferId == 0 ? PassDeclaration.DefaultFramebuffer : framebufferId, (uint)mask); + } + + /// glBlendEquationi + glBlendFuncSeparatei. + internal void StateSlotBlend(int slot, int equation, int srcColor, int dstColor, int srcAlpha, int dstAlpha) + { + stated.SetSlotBlend(slot, equation, srcColor, dstColor, srcAlpha, dstAlpha); + } + + /// glBlendFuncSeparatei alone: the attachment's equation stays. + internal void StateSlotBlendFunc(int slot, int srcColor, int dstColor, int srcAlpha, int dstAlpha) + { + stated.SetSlotFunc(slot, srcColor, dstColor, srcAlpha, dstAlpha); + } + + /// Blend on with a mode's functions on every attachment, or off with the functions kept. + internal void StateBlend(bool on, EnumBlendMode mode) + { + stated.SetBlendEnabled(on); + if (on) stated.SetBlendMode(mode); + } + + /// A viewport the client states (the fork bridge, and the platform's own full-target binds). + internal void NoteForkViewport(int x, int y, int width, int height) => + stated.Viewport = new Rect2D(new Offset2D(x, y), new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + + internal void NoteForkTexture(int unit, int textureId) => stated.BindTexture(unit, textureId); + + internal void NoteForkDrawBuffers(int framebufferId, int mask) => + stated.SetDrawBuffers(framebufferId == 0 ? PassDeclaration.DefaultFramebuffer : framebufferId, (uint)mask); + + /// The viewport the client last stated: what every native pass that keeps the viewport draws with. + private Rect2D StatedViewport() => stated.Viewport; + + /// The target the client's next draw or clear addresses: a fork's bound id, else CurrentFrameBuffer, else the default. + internal int CurrentTargetId => forkFramebuffer > 0 + ? forkFramebuffer + : CurrentFrameBuffer != null ? CurrentFrameBuffer.FboId : PassDeclaration.DefaultFramebuffer; + + /// A colour clear as OpenGL does it: only a selected draw buffer, never through an all-false colour mask. + internal void ClearTargetColor(int framebufferId, int slot, float r, float g, float b, float a) + { + if (framebufferId == 0) framebufferId = PassDeclaration.DefaultFramebuffer; + if (((stated.DrawBuffers(framebufferId) >> slot) & 1) == 0 || stated.ColorMask == 0) return; + device.ClearNativeColor(framebufferId, slot, r, g, b, a); + } + + /// A depth clear as OpenGL does it: not with depth writes off. + internal void ClearTargetDepth(int framebufferId, float depth) + { + if (framebufferId == 0) framebufferId = PassDeclaration.DefaultFramebuffer; + if (!stated.DepthWrite) return; + device.ClearNativeDepth(framebufferId, depth); + } + + // ------------------------------------------------------------------------- the route + + /// + /// Records one draw of the current program natively from the stated state. A null + /// is the fullscreen triangle; is a + /// pool's multi-draw. False: nothing was recorded (reported once per program). + /// + private bool TryDrawStated(VAO? vao, int instances, int[]? starts, int[]? sizes, int groupCount) + { + if (vao != null && (vao.VaoId == 0 || vao.Disposed)) return false; + return RecordStatedDraw(vao?.VaoId ?? 0, instances, starts, sizes, groupCount); + } + + /// The generic draw of a device mesh (0: the fullscreen triangle) into the current target. + internal bool RecordStatedDraw(int meshId, int instances, int[]? starts, int[]? sizes, int groupCount) + { + if (device == null || statedProgram <= 0) return false; + int programId = statedProgram; + + int framebufferId = CurrentTargetId; + RuntimeStats.drawCallsCount++; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + PassDeclaration? declared = statedPass != null && statedPass.FramebufferId == framebufferId ? statedPass : null; + bool drawn = StatedDraw.Record(device, stated, programId, framebufferId, + meshId, instances, starts, sizes, groupCount, out string? refusal, declared); + SetPassContext(outer, outerFlags); + if (drawn) + { + StatedDrawsForTests++; + return true; + } + RuntimeStats.drawCallsCount--; + return refusal == null ? false : Refuse(programId, refusal); + } + + private bool Refuse(int programId, string reason) + { + if (statedRefusalReported.Add(programId)) + { + ShaderProgramBase? current = ShaderProgramBase.CurrentShaderProgram; + string name = current != null && current.ProgramId == programId ? current.PassName ?? "" : "#" + programId; + Logger.Warning("Optimum: a draw of program '{0}' was dropped: {1}", name, reason); + } + return false; + } + +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Shaders.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Shaders.cs new file mode 100644 index 00000000..a394670d --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Shaders.cs @@ -0,0 +1,250 @@ +using System; +using System.Collections.Generic; +using OpenTK.Mathematics; +using Vintagestory.API.Client; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: shader compile and link, programs, uniforms, texture +// binding and uniform buffers. Each body is the device branch that opened the same +// ClientPlatformWindows method (Phase 1A step 3 moved the program/uniform/UBO ones there +// from ShaderProgramBase and UBO), moved unchanged. +// +// A uniform location here is whatever GetUniformLocation handed out - a byte offset into +// the generated block - so the setters read the same uniformLocations dictionary as GL. +// UBO.Handle carries the device's uniform-buffer handle, the convention VAO.VaoId uses for +// meshes, and the binding point survives from CreateUBO. +public partial class VulkanClientPlatform +{ + /// + /// The device resolves the block by name against the program's reflected interface and + /// binds it to the same point, so the four GL steps - allocate, look up the block index, + /// bind the index, bind the buffer - collapse into one call. + /// + public override UBORef CreateUBO(int shaderProgramId, int bindingPoint, string blockName, int size) + { + UBO optimumUbo = new UBO(); + optimumUbo.Handle = device.CreateUniformBuffer(shaderProgramId, bindingPoint, blockName, size); + optimumUbo.Size = size; + optimumUbo.BlockName = blockName; + optimumUbo.BindingPoint = bindingPoint; + return optimumUbo; + } + + public override void BindUBO(UBO ubo) + { + device.BindUniformBuffer(ubo.Handle); + } + + public override void UnbindUBO(UBO ubo) + { + device.UnbindUniformBuffer(ubo.Handle); + } + + /// The device writes a range whether or not the GL path would reallocate. + public override void UpdateUBO(UBO ubo, IntPtr data, int offset, int size, bool reallocate) + { + device.UpdateUniformBuffer(ubo.Handle, data, offset, size); + } + + public override void DeleteUBO(UBO ubo) + { + device.DeleteUniformBuffer(ubo.Handle); + } + + public override int GetUniformLocation(ShaderProgram program, string name) + { + return device.GetUniformLocation(program.ProgramId, name); + } + + /// + /// glUseProgram, recorded: the generic stated draw draws this program. The dedicated native + /// routes read ShaderProgramBase.CurrentShaderProgram, which ShaderProgramBase.Use and Stop set + /// right after this call, so the two never disagree in the client. + /// + public override void UseShaderProgram(int programId) + { + statedProgram = programId; + } + + /// + /// The device owns the SPIR-V modules inside the program and frees them with it, so + /// there is nothing matching GL's detach-and-delete of the individual stages. + /// + public override void DisposeShaderProgram(ShaderProgramBase program) + { + foreach (KeyValuePair optimumSampler in program.customSamplers) + { + device.DeleteSampler(optimumSampler.Value); + } + ForgetNativeChunkProgram(program.ProgramId); + device.DeleteProgram(program.ProgramId); + } + + public override void BindSampler(int unit, int samplerId) + { + stated.BindSampler(unit, samplerId); + } + + public override void SetUniform(int programId, int location, float value) + { + device.SetUniform(programId, location, value); + } + + public override void SetUniform(int programId, int location, int value) + { + device.SetUniform(programId, location, value); + } + + public override void SetUniform(int programId, int location, float x, float y) + { + device.SetUniform(programId, location, x, y); + } + + public override void SetUniform(int programId, int location, float x, float y, float z) + { + device.SetUniform(programId, location, x, y, z); + } + + public override void SetUniform(int programId, int location, float x, float y, float z, float w) + { + device.SetUniform(programId, location, x, y, z, w); + } + + public override void SetUniform(int programId, int location, int x, int y, int z) + { + // Unlike the Vec2i overload, which casts to float before it gets here, + // Vec3i keeps integers, so the shader declares an ivec3. The location is + // opaque to this side, so the device lays the three components out itself. + device.SetUniform(programId, location, x, y, z); + } + + public override void SetUniformArray1(int programId, int location, int count, float[] values) + { + device.SetUniformArray1(programId, location, count, values); + } + + public override void SetUniformArray2(int programId, int location, int count, float[] values) + { + device.SetUniformArray2(programId, location, count, values); + } + + public override void SetUniformArray3(int programId, int location, int count, float[] values) + { + device.SetUniformArray3(programId, location, count, values); + } + + public override void SetUniformArray4(int programId, int location, int count, float[] values) + { + device.SetUniformArray4(programId, location, count, values); + } + + public override void SetUniformMatrix(int programId, int location, float[] matrix) + { + device.SetUniformMatrix(programId, location, matrix); + } + + public override void SetUniformMatrix(int programId, int location, ref Matrix4 matrix) + { + // Only the sun and moon renderers use this overload, a handful of + // times per frame, so flattening into an array here is not worth a + // dedicated entry point on the seam. + float[] optimumMatrix = new float[16]; + optimumMatrix[0] = matrix.M11; optimumMatrix[1] = matrix.M12; + optimumMatrix[2] = matrix.M13; optimumMatrix[3] = matrix.M14; + optimumMatrix[4] = matrix.M21; optimumMatrix[5] = matrix.M22; + optimumMatrix[6] = matrix.M23; optimumMatrix[7] = matrix.M24; + optimumMatrix[8] = matrix.M31; optimumMatrix[9] = matrix.M32; + optimumMatrix[10] = matrix.M33; optimumMatrix[11] = matrix.M34; + optimumMatrix[12] = matrix.M41; optimumMatrix[13] = matrix.M42; + optimumMatrix[14] = matrix.M43; optimumMatrix[15] = matrix.M44; + device.SetUniformMatrix(programId, location, optimumMatrix); + } + + public override void SetUniformMatrices(int programId, int location, int count, float[] matrices) + { + device.SetUniformMatrices(programId, location, count, matrices); + } + + public override void SetUniformMatrices4x3(int programId, int location, int count, float[] matrices) + { + device.SetUniformMatrices4x3(programId, location, count, matrices); + } + + /// + /// In GL this is three separate things - point the sampler uniform at a unit, activate + /// that unit, bind the texture. The device keeps the same split so the two halves can + /// be set independently, which the render systems rely on. + /// + public override void BindProgramTexture2D(ShaderProgramBase program, string samplerName, int textureId, int textureNumber) + { + // Phase 3b stage 2: this is where the client states which texture a sampler reads - + // "this program's sampler is this texture" - so it is where a native pass takes + // the handle from (VulkanClientPlatform.NativeChunks.cs, .NativeEntities.cs). It is not + // the device's texture-unit table: the unit binding is recorded in the platform's stated + // state, which the generic native draw resolves samplers through. + NoteNativeProgramTexture(program.ProgramId, samplerName, textureId); + device.SetSamplerUnit(program.ProgramId, samplerName, textureNumber); + stated.BindTexture(textureNumber, textureId); + if (program.customSamplers.TryGetValue(samplerName, out var optimumSampler)) + { + stated.BindSampler(textureNumber, optimumSampler); + } + else + { + // Clear any override left on this unit, or the texture's own + // filtering would be silently ignored. + stated.BindSampler(textureNumber, 0); + } + if (program.clampTToEdge) + { + device.SetTextureParameter(textureId, + Vintagestory.API.Config.OptimumGlConstants.TextureWrapT, + Vintagestory.API.Config.OptimumGlConstants.ClampToEdge); + } + } + + public override void BindProgramTextureCube(ShaderProgramBase program, string samplerName, int textureId, int textureNumber) + { + NoteNativeProgramTexture(program.ProgramId, samplerName, textureId); + device.SetSamplerUnit(program.ProgramId, samplerName, textureNumber); + stated.BindTexture(textureNumber, textureId); + if (program.clampTToEdge) + { + device.SetTextureParameter(textureId, + Vintagestory.API.Config.OptimumGlConstants.TextureWrapT, + Vintagestory.API.Config.OptimumGlConstants.ClampToEdge); + } + } + + /// + /// The device only stages the stage here. GL resolves uniforms and varyings by name + /// across the whole program, so nothing about a stage is final until its siblings are + /// known, and the real translation happens at link time. + /// + public override bool CompileShader(Shader shader) + { + return device.CompileShader(shader); + } + + /// + /// The device assigns the id, exactly as glCreateProgram did, and the caller stores it - + /// IShaderProgram.ProgramId is read-only on the interface, so it comes back as a + /// return value. + /// + public override bool CreateShaderProgram(ShaderProgram program) + { + int optimumProgramId = device.LinkProgram(program); + if (optimumProgramId == 0) + { + string optimumLinkError = device.GetError(); + Logger.Error("Link error in shader program for pass {0}: {1}", + program.PassName, optimumLinkError == null ? "unknown" : optimumLinkError); + return false; + } + program.ProgramId = optimumProgramId; + Logger.Notification("Loaded Shaderprogramm for render pass {0}.", program.PassName); + return true; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.State.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.State.cs new file mode 100644 index 00000000..26b3af66 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.State.cs @@ -0,0 +1,330 @@ +using Optimum.Render.Vulkan.Core; +using Silk.NET.Vulkan; +using System; +using System.Runtime.InteropServices; +using Cairo; +using OpenTK.Audio.OpenAL; +using Vintagestory.API.Client; +using Vintagestory.Client.NoObf; +using Vintagestory.Common.Convert; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: fixed-function state, diagnostics and capability +// reporting. Each body is the device branch that used to open the same method in +// ClientPlatformWindows, moved unchanged; ClientPlatformWindows keeps the GL body. +public partial class VulkanClientPlatform +{ + // The device takes these as call arguments and keeps no queryable state, so the + // platform remembers what the GL driver would have: the debug flag for its getter, + // the scissor flag the runtime atlas upload reads back, and the texture last bound to + // unit 0 for the argument-less GlGenerateTex2DMipmaps. + private bool debugMode; + private bool scissorEnabled; + private int boundTexture2d; + + public override bool GlDebugMode + { + get + { + return debugMode; + } + set + { + // The device's equivalent is the validation layer, which it enables + // itself; the supportsGlDebugMode check does not apply. + device.DebugMode = value; + debugMode = value; + } + } + + public override bool GlScissorFlagEnabled + { + get + { + return scissorEnabled; + } + } + + public override string GetGraphicsCardRenderer() + { + return device.RendererString; + } + + /// + /// The same facts the GL body logs, from the device. The GL extension test has no + /// meaning here: the device already refused to initialize if it lacked what it needs, + /// and supportsGlDebugMode/supportsPersistentMapping stay false because only GL + /// bodies read them. + /// + public override void LogAndTestHardwareInfosStage2() + { + Logger.Notification("Graphics Backend: " + device.BackendName); + Logger.Notification("Graphics Card Vendor: " + device.VendorString); + Logger.Notification("Graphics Card Version: " + device.VersionString); + Logger.Notification("Graphics Card Renderer: " + device.RendererString); + Logger.Notification("Graphics Card ShadingLanguageVersion: " + device.ShaderVersionString); + Logger.Notification("Max texture size: " + device.MaxTextureSize); + if (!RuntimeInformation.IsOSPlatform(OSPlatform.OSX)) + { + // ClientPlatformWindows.LogFrameworkVersions, which is private. + Logger.Notification("C# Framework: " + GetFrameworkInfos()); + Logger.Notification("Cairo Graphics Version: " + CairoAPI.VersionString); + } + Logger.Notification("OpenAL Version: " + AL.Get((ALGetString)45058)); + Logger.Notification("Zstd Version: " + ZstdNative.Version); + CheckGlError("loghwinfo"); + } + + /// + /// This text goes into crash reports, so it names the backend too - a crash on + /// Vulkan reads very differently from the same crash on GL. + /// + public override string GetGraphicCardInfos() + { + return "GC Backend: " + device.BackendName + "\nGC Vendor: " + device.VendorString + "\nGC Version: " + device.VersionString + "\nGC Renderer: " + device.RendererString + "\nGC ShaderVersion: " + device.ShaderVersionString; + } + + /// The device drains validation-layer messages instead of a GL error code. + public override void CheckGlError(string errmsg = null) + { + if (GlErrorChecking) + { + string optimumError = device.GetError(); + if (optimumError != null) + { + throw new Exception(((errmsg == null) ? "" : (errmsg + " ")) + "- the graphics backend reported: " + optimumError); + } + } + } + + public override void CheckGlErrorAlways(string errmsg = null) + { + string optimumError = device.GetError(); + if (optimumError != null) + { + Logger.Error(((errmsg == null) ? "" : (errmsg + " ")) + "- the graphics backend reported: " + optimumError); + } + } + + public override string GlGetError() + { + return device.GetError(); + } + + public override string GetGLShaderVersionString() + { + // The client parses this to decide whether a shader's #version is + // supported, so it has to keep reading as a GLSL version number. + return device.ShaderVersionString; + } + + public override int GenSampler(bool linear) + { + return device.CreateSampler(linear); + } + + public override void GLWireframes(bool toggle) + { + stated.Wireframe = toggle; + } + + public override void GlViewport(int x, int y, int width, int height) + { + stated.Viewport = new Rect2D(new Offset2D(x, y), new Extent2D((uint)Math.Max(0, width), (uint)Math.Max(0, height))); + } + + public override void GlScissor(int x, int y, int width, int height) + { + // Clipped to the positive quadrant (Vulkan rejects a negative offset; GL keeps the visible + // remainder), kept as client state for native passes. + int clippedX = Math.Max(0, x); + int clippedY = Math.Max(0, y); + statedScissor = new Rect2D(new Offset2D(clippedX, clippedY), + new Extent2D((uint)Math.Max(0, width - (clippedX - x)), (uint)Math.Max(0, height - (clippedY - y)))); + stated.Scissor = statedScissor; + } + + // The fixed state the client last stated through this platform's own virtuals. A native pass + // that draws "with whatever the caller set" (the GUI quads) reads these - client statements, + // recorded where they are made - never the device's GL state tracker (Phase 3b decision 3). + private bool statedBlendOn; + private EnumBlendMode statedBlendMode = EnumBlendMode.Standard; + private bool statedDepthTest; + private bool statedDepthWrite = true; + private int statedDepthFunc = 513; // GL_LESS + private Rect2D statedScissor; + private bool statedCull; + private bool statedCullBack = true; + private float statedLineWidth = 1f; + + public override void GlScissorFlag(bool enable) + { + scissorEnabled = enable; + stated.ScissorEnabled = enable; + } + + public override void GlEnableDepthTest() + { + statedDepthTest = true; + stated.DepthTest = true; + } + + public override void GlDisableDepthTest() + { + statedDepthTest = false; + stated.DepthTest = false; + } + + public override void BindTexture2d(int texture) + { + // The GL body activates unit 0 first, so this binds to unit 0 too. + stated.BindTexture(0, texture); + // Remembered for GlGenerateTex2DMipmaps, whose GL form acts on + // whatever is bound and so has no argument to route. + boundTexture2d = texture; + } + + public override void BindTextureCubeMap(int texture) + { + stated.BindTexture(0, texture); + } + + public override void UnBindTextureCubeMap() + { + // Mirrors BindTextureCubeMap above, which binds to unit 0. + stated.BindTexture(0, 0); + } + + public override void GlToggleBlend(bool on, EnumBlendMode blendMode = EnumBlendMode.Standard) + { + statedBlendOn = on; + statedBlendMode = blendMode; + // GL: glEnable(GL_BLEND) and the mode's functions on every draw buffer when on, + // glDisable alone - the functions stay - when off (ClientPlatformWindows.GlToggleBlend). + stated.SetBlendEnabled(on); + if (on) + { + stated.SetBlendMode(blendMode); + if (blendMode == EnumBlendMode.Standard && OptimumRenderSsao) + { + stated.SetSlotBlend(2, 32774, 1, 0, 1, 0); + stated.SetSlotBlend(3, 32774, 1, 0, 1, 0); + } + } + // Optimum TAA (P3): the motion attachment never blends. A blended + // motion vector averages two surfaces' displacements and belongs to + // neither; the per-attachment override has to be re-applied after + // every global blend change, exactly like the SSAO one above. + if (on) + { + ApplyOptimumMotionBlendState(); + } + } + + public override void GlDisableCullFace() + { + statedCull = false; + stated.CullEnabled = false; + } + + public override void GlEnableCullFace() + { + statedCull = true; + stated.CullEnabled = true; + } + + public override void GLLineWidth(float width) + { + statedLineWidth = width; + stated.LineWidth = width; + } + + /// + /// GL_LINE_SMOOTH has no Vulkan equivalent - smooth lines there are a + /// rasterization-mode on the pipeline, not toggleable state - and it is + /// purely cosmetic, so the device path ignores it rather than pretending. + /// + public override void SmoothLines(bool on) + { + } + + public override void GlDepthMask(bool flag) + { + statedDepthWrite = flag; + stated.DepthWrite = flag; + } + + public override void GlDepthFunc(EnumDepthFunction depthFunc) + { + // EnumDepthFunction's values are the GL constants, which is the form + // the seam takes: it cannot reference this enum, since it lives in + // VintagestoryLib and the contracts assembly does not depend on it. + statedDepthFunc = (int)depthFunc; + stated.DepthCompare = GlEnums.CompareOpFrom((int)depthFunc); + } + + public override void GlCullFaceBack() + { + statedCullBack = true; + stated.CullBack = true; + } + + public override void GlCullFaceFront() + { + statedCullBack = false; + stated.CullBack = false; + } + + public override void GlEnableStencilTest() + { + stated.StencilTest = true; + } + + public override void GlDisableStencilTest() + { + stated.StencilTest = false; + } + + public override void GlStencilMask(int mask) + { + } + + public override void GlStencilFunc(int func, int refVal, int mask) + { + } + + public override void GlStencilOp(int sfail, int dpfail, int dppass) + { + } + + /// + /// The colour channels the client last masked off with GlColorMask (None = all written), + /// stated for native draws: the sun's occlusion probe draws with every channel off. + /// + private ColorComponentFlags statedColorMaskOff; + + public override void GlColorMask(bool r, bool g, bool b, bool a) + { + statedColorMaskOff = (r ? 0 : ColorComponentFlags.RBit) | (g ? 0 : ColorComponentFlags.GBit) | + (b ? 0 : ColorComponentFlags.BBit) | (a ? 0 : ColorComponentFlags.ABit); + stated.SetColorMask(r, g, b, a); + } + + public override void GlClearStencil() + { + } + + public override void GlGenerateTex2DMipmaps() + { + if (boundTexture2d != 0) + { + device.GenerateMipmaps(boundTexture2d); + } + } + + public override int GlGetMaxTextureSize() + { + return device.MaxTextureSize; + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Textures.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Textures.cs new file mode 100644 index 00000000..87e1c739 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Textures.cs @@ -0,0 +1,294 @@ +using System; +using System.Runtime.InteropServices; +using Cairo; +using Vintagestory.API.Client; +using Vintagestory.API.Common; +using Vintagestory.API.Config; +using Vintagestory.ClientNative; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +// Vulkan-native plan, Phase 1A step 4: texture creation, upload and mipmapping. Each body is +// the device branch that opened the same ClientPlatformWindows method, moved unchanged, +// behind the same main-thread check. +public partial class VulkanClientPlatform +{ + private const string MainThreadOnly = "Texture uploads must happen in the main thread. We only have one OpenGL context."; + + /// + /// Cairo hands over premultiplied BGRA bytes, which is why the GL body asks for GL_BGRA + /// rather than GL_RGBA; CreateTexture2DRaw takes the same GL internal format token so + /// the device makes the identical image. + /// + public override int LoadCairoTexture(ImageSurface surface, bool linearMag) + { + if (Environment.CurrentManagedThreadId != RuntimeEnv.MainThreadId) + { + throw new InvalidOperationException(MainThreadOnly); + } + int optimumTextureId = device.CreateTexture2DRaw(surface.Width, surface.Height, + OptimumGlConstants.Bgra, surface.DataPtr, 4); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMagFilter, linearMag ? 9729 : 9728); + return optimumTextureId; + } + + public override void GenTexture(RawTexture tex) + { + if (Environment.CurrentManagedThreadId != RuntimeEnv.MainThreadId) + { + throw new InvalidOperationException(MainThreadOnly); + } + int optimumTextureId = device.CreateTexture2D(tex.Width, tex.Height, + tex.PixelInternalFormat, tex.PixelFormat, IntPtr.Zero, false); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMinFilter, (int)tex.MinFilter); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMagFilter, (int)tex.MagFilter); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapS, (int)tex.WrapS); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapT, (int)tex.WrapT); + tex.TextureId = optimumTextureId; + } + + public override void LoadOrUpdateCairoTexture(ImageSurface surface, bool linearMag, ref LoadedTexture intoTexture) + { + if (Environment.CurrentManagedThreadId != RuntimeEnv.MainThreadId) + { + throw new InvalidOperationException(MainThreadOnly); + } + if (intoTexture.TextureId == 0 || intoTexture.Width != surface.Width || intoTexture.Height != surface.Height) + { + if (intoTexture.TextureId != 0) + { + device.DeleteTexture(intoTexture.TextureId); + } + intoTexture.TextureId = device.CreateTexture2DRaw(surface.Width, surface.Height, + OptimumGlConstants.Bgra, surface.DataPtr, 4); + intoTexture.Width = surface.Width; + intoTexture.Height = surface.Height; + device.SetTextureParameter(intoTexture.TextureId, OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(intoTexture.TextureId, OptimumGlConstants.TextureMagFilter, linearMag ? 9729 : 9728); + } + else + { + // The image is BGRA-ordered; the upload is a byte copy at four + // bytes per pixel, which is what Rgba selects here. + device.UploadTexture2D(intoTexture.TextureId, 0, 0, 0, + surface.Width, surface.Height, EnumTexturePixelFormat.Rgba, surface.DataPtr); + } + CheckGlError("LoadOrUpdateCairoTexture"); + } + + public override unsafe void LoadIntoTexture(IBitmap srcBmp, int targetTextureId, int destX, int destY, bool generateMipmaps = false) + { + if (srcBmp is BitmapExternal optimumExternal) + { + device.UploadTexture2D(targetTextureId, 0, destX, destY, + srcBmp.Width, srcBmp.Height, EnumTexturePixelFormat.Rgba, + (IntPtr)optimumExternal.PixelsPtrAndLock); + } + else + { + // A managed pixel array has to be pinned before the device can + // read it; the GL body relied on the overload doing that. + GCHandle optimumPin = GCHandle.Alloc(srcBmp.Pixels, GCHandleType.Pinned); + try + { + device.UploadTexture2D(targetTextureId, 0, destX, destY, + srcBmp.Width, srcBmp.Height, EnumTexturePixelFormat.Rgba, + optimumPin.AddrOfPinnedObject()); + } + finally + { + optimumPin.Free(); + } + } + if (ENABLE_MIPMAPS && generateMipmaps) + { + BuildMipMaps(targetTextureId); + } + } + + /// + /// The GL body uploads BGRA bytes into a GL_RGBA image; the device gets a BGRA-ordered + /// image instead, which samples the same way without a per-pixel swizzle. Anisotropy is + /// a sampler property the device sets from its own limit, so there is nothing to query. + /// + public override unsafe int LoadTexture(IBitmap bmp, bool linearMag = false, int clampMode = 0, bool generateMipmaps = false) + { + if (Environment.CurrentManagedThreadId != RuntimeEnv.MainThreadId) + { + throw new InvalidOperationException(MainThreadOnly); + } + int optimumTextureId; + if (bmp is BitmapExternal optimumExternal) + { + optimumTextureId = device.CreateTexture2DRaw(bmp.Width, bmp.Height, + OptimumGlConstants.Bgra, (IntPtr)optimumExternal.PixelsPtrAndLock, 4, + ENABLE_MIPMAPS && generateMipmaps); + } + else + { + GCHandle optimumPin = GCHandle.Alloc(bmp.Pixels, GCHandleType.Pinned); + try + { + optimumTextureId = device.CreateTexture2DRaw(bmp.Width, bmp.Height, + OptimumGlConstants.Bgra, optimumPin.AddrOfPinnedObject(), 4, + ENABLE_MIPMAPS && generateMipmaps); + } + finally + { + optimumPin.Free(); + } + } + switch (clampMode) + { + case 1: + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapS, 33071); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapT, 33071); + break; + case 2: + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapS, 10497); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureWrapT, 10497); + break; + } + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(optimumTextureId, OptimumGlConstants.TextureMagFilter, linearMag ? 9729 : 9728); + if (ENABLE_MIPMAPS && generateMipmaps) + { + BuildMipMaps(optimumTextureId); + } + return optimumTextureId; + } + + public override void LoadOrUpdateTextureFromBgra_DeferMipMap(int[] rgbaPixels, bool linearMag, int clampMode, ref LoadedTexture intoTexture) + { + LoadOrUpdateTextureFromPixels(rgbaPixels, linearMag, clampMode, ref intoTexture, bgra: true, makeMipMap: false); + } + + public override void LoadOrUpdateTextureFromBgra(int[] rgbaPixels, bool linearMag, int clampMode, ref LoadedTexture intoTexture) + { + LoadOrUpdateTextureFromPixels(rgbaPixels, linearMag, clampMode, ref intoTexture, bgra: true, makeMipMap: true); + } + + public override void LoadOrUpdateTextureFromRgba(int[] rgbaPixels, bool linearMag, int clampMode, ref LoadedTexture intoTexture) + { + LoadOrUpdateTextureFromPixels(rgbaPixels, linearMag, clampMode, ref intoTexture, bgra: false, makeMipMap: true); + } + + /// + /// The texture atlas upload path: TextureAtlas.Upload reaches it through + /// LoadOrUpdateTextureFromBgra_DeferMipMap, which is why it only runs once a world + /// starts loading and never on the menu. The pixels are BGRA when + /// says so (the GL body's PixelFormat 32993) and RGBA otherwise, matching the wrappers. + /// + private void LoadOrUpdateTextureFromPixels(int[] rgbaPixels, bool linearMag, int clampMode, ref LoadedTexture intoTexture, bool bgra, bool makeMipMap) + { + if (Environment.CurrentManagedThreadId != RuntimeEnv.MainThreadId) + { + throw new InvalidOperationException(MainThreadOnly); + } + int optimumGlFormat = bgra + ? Vintagestory.API.Config.OptimumGlConstants.Bgra + : Vintagestory.API.Config.OptimumGlConstants.Rgba8; + + GCHandle optimumPin = GCHandle.Alloc(rgbaPixels, GCHandleType.Pinned); + try + { + if (intoTexture.TextureId == 0 || intoTexture.Width * intoTexture.Height != rgbaPixels.Length) + { + if (intoTexture.TextureId != 0) + { + device.DeleteTexture(intoTexture.TextureId); + } + // The mip chain has to be requested at creation; asking for + // mipmaps afterwards on a one-level image does nothing. GL + // can grow one at any time, which is what the deferred + // variant relies on: it uploads with makeMipMap false and + // the atlas manager calls BuildMipMaps later, in StageB. So + // the chain is sized whenever mipmapping is on at all, and + // makeMipMap only decides whether to fill it here. + intoTexture.TextureId = device.CreateTexture2DRaw( + intoTexture.Width, intoTexture.Height, optimumGlFormat, + optimumPin.AddrOfPinnedObject(), 4, ENABLE_MIPMAPS); + + if (clampMode == 1) + { + device.SetTextureParameter(intoTexture.TextureId, + Vintagestory.API.Config.OptimumGlConstants.TextureWrapS, 33071); + device.SetTextureParameter(intoTexture.TextureId, + Vintagestory.API.Config.OptimumGlConstants.TextureWrapT, 33071); + } + device.SetTextureParameter(intoTexture.TextureId, + Vintagestory.API.Config.OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(intoTexture.TextureId, + Vintagestory.API.Config.OptimumGlConstants.TextureMagFilter, linearMag ? 9729 : 9728); + + if (makeMipMap) + { + BuildMipMaps(intoTexture.TextureId); + } + } + else + { + device.UploadTexture2D(intoTexture.TextureId, 0, 0, 0, + intoTexture.Width, intoTexture.Height, + EnumTexturePixelFormat.Rgba, optimumPin.AddrOfPinnedObject()); + } + } + finally + { + optimumPin.Free(); + } + } + + /// + /// The device sizes the mip chain when the image is created, so the generate carries + /// over as it is. The two glTexParameter calls carry over as well, and they are not + /// decoration: GL_LINEAR means "level 0 only" whatever the chain holds, so a texture + /// that is never moved to a MIPMAP filter is never minified through one. Vulkan has no + /// such filter, and the device turns these two into the sampler's LOD clamp. + /// + public override void BuildMipMaps(int textureId) + { + if (ENABLE_MIPMAPS) + { + device.GenerateMipmaps(textureId); + device.SetTextureParameter(textureId, + Vintagestory.API.Config.OptimumGlConstants.TextureMinFilter, 9986); + device.SetTextureParameter(textureId, + Vintagestory.API.Config.OptimumGlConstants.TextureMaxLevel, ClientSettings.MipMapLevel); + } + } + + /// + /// The skybox cubemap. CreateTextureCube takes all six faces at once, so the per-side + /// helper the GL body calls has no counterpart here. + /// + public override unsafe int Load3DTextureCube(BitmapRef[] bmps) + { + IntPtr[] optimumFaces = new IntPtr[6]; + int optimumSize = 0; + for (int k = 0; k < 6; k++) + { + BitmapExternal optimumFace = (BitmapExternal)bmps[k]; + optimumSize = optimumFace.Width; + optimumFaces[k] = (IntPtr)optimumFace.PixelsPtrAndLock; + } + // BGRA like the other bitmap uploads, so the raw overload rather than + // the EnumTextureInternalFormat one. + int optimumCubeId = device.CreateTextureCubeRaw(optimumSize, + OptimumGlConstants.Bgra, optimumFaces, 4); + device.SetTextureParameter(optimumCubeId, OptimumGlConstants.TextureMinFilter, 9729); + device.SetTextureParameter(optimumCubeId, OptimumGlConstants.TextureMagFilter, 9729); + device.SetTextureParameter(optimumCubeId, OptimumGlConstants.TextureWrapS, 33071); + device.SetTextureParameter(optimumCubeId, OptimumGlConstants.TextureWrapT, 33071); + return optimumCubeId; + } + + public override void GLDeleteTexture(int id) + { + // The device defers the destruction until the GPU is finished with + // the frame that referenced it; GL left that to the driver. + device.DeleteTexture(id); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.UiSeparation.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.UiSeparation.cs new file mode 100644 index 00000000..a9cb296e --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.UiSeparation.cs @@ -0,0 +1,283 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Vintagestory.API.Client; +using Vintagestory.API.MathTools; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +/// +/// World/UI separation, a foundation of the Vulkan renderer (owner's call, 2026-09-17: always on, +/// no switch). The frame is rendered HUD-less and the UI is composed onto it at the end - the +/// structure every engine with an upscaler or a frame generator needs, because the UI must never +/// enter an image a reconstruction consumes. OpenGL keeps drawing its GUI straight onto the window. +/// +/// Two images, both published by slot the way MotionAttachmentIndex is: +/// +/// SceneNoHud (slot 23, render size, RGBA8): a copy of Primary colour 0 +/// taken at the end of - the one moment the composited image +/// holds the scene alone. RenderAfterFinalComposition draws selection boxes and work-item guides onto +/// that image next, and the GUI follows after the blit. +/// UI image (slot 24, window size, RGBA8 + depth): everything after the +/// blit - the AfterBlit stage, the main-menu background and the whole Ortho stage - draws into it +/// instead of onto the window, over transparent black, with real coverage in alpha (see +/// ). Its own depth, because the GUI depth-sorts itself over +/// ScreenManager's 0..20000 range. +/// +/// +/// The UI scope runs from (the end of ) +/// to (ClientMain before its Done stage; ScreenManager for the +/// menu screens). While it is open the device resolves Default to the UI image +/// (), so no render system needs to know: what +/// has always drawn "onto the window" draws into the UI image, the atlas item renderer's +/// LoadFrameBuffer(Default) included. The compose closes the scope first and is the only draw that +/// writes the window image after the blit. and every framebuffer rebuild +/// close a scope a throwing GUI renderer left open. +/// +/// The OpenGL bodies this mirrors are the feat/dlss-g ones (design step 3); there the +/// target was gated on frame generation, here it is not gated. Pinned by UiSeparationTests (GPU) +/// and ui-separation-coverage-tests.cs (placement). +/// +public partial class VulkanClientPlatform +{ + /// The HUD-less scene snapshot's slot (named in ClientPlatformWindows' parity dump). + internal const int OptimumSceneNoHudIndex = 23; + + /// The UI image's slot (named in ClientPlatformWindows' parity dump). + internal const int OptimumUiTargetIndex = 24; + + /// + /// The slot holding this frame's HUD-less scene, or -1 when it could not be allocated. Consumers + /// ask this before they index FrameBuffers. + /// + public int SceneNoHudFrameBufferIndex => sceneNoHudIndex; + + /// The slot holding this frame's UI image, or -1 when it could not be allocated. + public int UiTargetFrameBufferIndex => uiTargetIndex; + + private int sceneNoHudIndex = -1; + private int uiTargetIndex = -1; + + /// + /// Whether the last composition really wrote the snapshot. A consumer that reads the slot when + /// this is false reads an earlier frame's image. + /// + public bool SceneNoHudCaptured { get; private set; } + + /// True from to . + internal bool UiScopeOpen => stated.UiImageFramebuffer > 0; + + /// The snapshot copy: the pass-through program, opaque, render size. + private readonly NativeFullscreenPass nativeSceneNoHudCopy = new("ui-compose", Array.Empty(), new[] { "uiTex" }); + + /// The compose: the pass-through program under premultiplied-alpha blending. + private readonly NativeFullscreenPass nativeUiCompose = new("ui-compose", Array.Empty(), new[] { "uiTex" }); + + private static AttachmentBlend[] PremultipliedColorZero() => + new[] { AttachmentBlend.For(true, EnumBlendMode.PremultipliedAlpha) }; + + /// + /// Allocates both images into and publishes their slots. A failure costs + /// that image and nothing else: its slot stays -1, the snapshot is simply not taken, and without a + /// UI image the GUI draws straight onto the window as it does on OpenGL. + /// + internal void AllocateUiSeparationTargets(List list, int renderWidth, int renderHeight) + { + CloseUiScope(); + sceneNoHudIndex = -1; + uiTargetIndex = -1; + SceneNoHudCaptured = false; + + try + { + FrameBufferRef snapshot = CreateOptimumOwnedTarget(renderWidth, renderHeight, withDepth: false); + list[OptimumSceneNoHudIndex] = snapshot; + sceneNoHudIndex = OptimumSceneNoHudIndex; + } + catch (Exception error) + { + Logger.Error("Optimum: no HUD-less scene snapshot: {0}", error.Message); + list[OptimumSceneNoHudIndex] = null; + } + + // The window's size, not the render size: the GUI has always laid itself out in window + // pixels, and the compose puts it back over the window one for one. + Size2i window = OptimumWindowClientSize(); + int windowWidth = window.Width; + int windowHeight = window.Height; + try + { + FrameBufferRef ui = CreateOptimumOwnedTarget(windowWidth, windowHeight, withDepth: true); + list[OptimumUiTargetIndex] = ui; + uiTargetIndex = OptimumUiTargetIndex; + } + catch (Exception error) + { + Logger.Error("Optimum: no separate UI image, the GUI draws onto the window: {0}", error.Message); + list[OptimumUiTargetIndex] = null; + } + } + + /// + /// A persistent single-colour target (RGBA8, nearest, clamped: both images are read one texel for + /// one), with a depth attachment when asked. Not a transient: its contents outlive the pass that + /// wrote them. + /// + private FrameBufferRef CreateOptimumOwnedTarget(int width, int height, bool withDepth) + { + FrameBufferRef target = new FrameBufferRef(); + target.Width = width; + target.Height = height; + target.FboId = device.CreateFramebuffer(width, height); + target.ColorTextureIds = new int[1]; + target.ColorTextureIds[0] = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.Rgba8, EnumTexturePixelFormat.Rgba, IntPtr.Zero, false); + SetupOptimumTextureSampler(target.ColorTextureIds[0], 9728, 33071); + device.AttachTexture(target.FboId, EnumFramebufferAttachment.ColorAttachment0, target.ColorTextureIds[0], 0); + if (withDepth) + { + target.DepthTextureId = device.CreateTexture2D(width, height, + EnumTextureInternalFormat.DepthComponent32, EnumTexturePixelFormat.DepthComponent, IntPtr.Zero, false); + SetupOptimumTextureSampler(target.DepthTextureId, 9728, 33071); + device.AttachTexture(target.FboId, EnumFramebufferAttachment.DepthAttachment, target.DepthTextureId, 0); + } + StateDrawBuffers(target.FboId, 1); + if (!device.CheckFramebufferComplete(target.FboId, out string status)) + { + throw new Exception("framebuffer incomplete: " + status); + } + return target; + } + + /// + /// Takes the HUD-less snapshot: Primary colour 0 copied texel for texel into slot 23, one native + /// pass. Called at the very end of , whichever route drew it. + /// + internal void CaptureSceneNoHud() + { + SceneNoHudCaptured = false; + List buffers = FrameBuffers; + if (sceneNoHudIndex < 0 || buffers == null || buffers.Count <= sceneNoHudIndex) return; + FrameBufferRef snapshot = buffers[sceneNoHudIndex]; + FrameBufferRef primary = buffers[0]; + if (snapshot == null || primary?.ColorTextureIds == null || primary.ColorTextureIds.Length == 0) return; + ShaderProgram copy = ShaderPrograms.UiCompose; + if (copy == null || copy.LoadError || copy.ProgramId <= 0) return; + + int scene = primary.ColorTextureIds[0]; + string outer = passContext; + PassFlags outerFlags = passContextFlags; + NativePipeline? pipeline = NativePipelineFor(nativeSceneNoHudCopy, copy, snapshot.FboId); + if (pipeline != null && + BeginNativeBlitPass("SceneNoHud/" + OptimumSceneNoHudIndex, snapshot.FboId, + snapshot.Width, snapshot.Height, new[] { scene })) + { + SceneNoHudCaptured = device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeSceneNoHudCopy.Samplers[0], scene), + }); + } + device.EndNativePass(); + SetPassContext(outer, outerFlags); + } + + /// + /// Opens the UI scope: the UI image is cleared to transparent black and depth 1, and Default + /// resolves to it until the compose. Called at the end of on + /// every route out of it, the menu screens' included - the blit is the boundary between the world + /// and everything ScreenManager draws after it. The client still sees the Default target bound. + /// + internal void OpenUiScope() + { + CloseUiScope(); + List buffers = FrameBuffers; + if (uiTargetIndex < 0 || buffers == null || buffers.Count <= uiTargetIndex) return; + FrameBufferRef ui = buffers[uiTargetIndex]; + if (ui == null || ui.ColorTextureIds == null || ui.ColorTextureIds.Length == 0) return; + + // Straight to the device: the image has to start empty whatever colour or depth mask the + // client stated last, which the stated clears would honour. + device.ClearNativeColor(ui.FboId, 0, 0f, 0f, 0f, 0f); + device.ClearNativeDepth(ui.FboId, 1f); + device.RedirectDefaultFramebuffer(ui.FboId); + stated.UiImageFramebuffer = ui.FboId; + } + + /// Ends the scope without composing: Default is the window again and Standard is Standard. + internal void CloseUiScope() + { + stated.UiImageFramebuffer = 0; + device?.RedirectDefaultFramebuffer(0); + } + + /// + /// Puts the UI image back over the window image: one fullscreen pass under premultiplied-alpha + /// blending, dst = ui.rgb + dst.rgb * (1 - ui.a). The OpenGL body (feat/dlss-g) is the same + /// sequence against GL state. + /// + /// Where it is called from, and why there: ClientMain.RenderToDefaultFramebuffer after the Ortho + /// stage and before TriggerRenderStage(Done), where the with-HUD screenshot and the AVI writer are + /// registered, and ScreenManager.Render for the menu screens, which never reach ClientMain. A + /// second call in a frame finds the scope closed and returns. + /// + public override void OptimumComposeUiTarget() + { + if (!UiScopeOpen) return; + FrameBufferRef ui = UiTargetFrameBuffer; + // The scope ends here whatever follows, so a compose that gives up below never leaves Default + // pointing at the UI image or the separate-alpha rule armed for the rest of the frame. + CloseUiScope(); + // The window, bound and with its own viewport, whether or not the compose happens: the Done + // stage and the screenshot read it. + LoadFrameBuffer(EnumFrameBuffer.Default); + // ScreenManager's ClearDefaultDepth landed on the UI image, so the window's depth never got + // this frame's clear; vanilla hands the Done stage a freshly cleared one. + device.ClearNativeDepth(PassDeclaration.DefaultFramebuffer, 1f); + GlToggleBlend(true); + + ShaderProgram compose = ShaderPrograms.UiCompose; + if (ui == null || ui.ColorTextureIds == null || ui.ColorTextureIds.Length == 0) return; + if (compose == null || compose.LoadError || compose.ProgramId <= 0) return; + + int uiColor = ui.ColorTextureIds[0]; + Size2i client = OptimumWindowClientSize(); + string outer = passContext; + PassFlags outerFlags = passContextFlags; + NativePipeline? pipeline = NativePipelineFor(nativeUiCompose, compose, NativeDefaultTarget, + PremultipliedColorZero()); + if (pipeline != null && + BeginNativeBlitPass("UiCompose/Default", NativeDefaultTarget, client.Width, client.Height, new[] { uiColor })) + { + device.DrawNativeFullscreen(pipeline, new[] + { + new NativeTexture(nativeUiCompose.Samplers[0], uiColor), + }); + } + device.EndNativePass(); + SetPassContext(outer, outerFlags); + } + + /// The UI image, or null when its slot is not allocated. + internal FrameBufferRef? UiTargetFrameBuffer + { + get + { + List buffers = FrameBuffers; + if (uiTargetIndex < 0 || buffers == null || buffers.Count <= uiTargetIndex) return null; + return buffers[uiTargetIndex]; + } + } + + /// The HUD-less scene snapshot, or null when its slot is not allocated. + internal FrameBufferRef? SceneNoHudFrameBuffer + { + get + { + List buffers = FrameBuffers; + if (sceneNoHudIndex < 0 || buffers == null || buffers.Count <= sceneNoHudIndex) return null; + return buffers[sceneNoHudIndex]; + } + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.cs b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.cs new file mode 100644 index 00000000..4c4ceba3 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanClientPlatform.cs @@ -0,0 +1,385 @@ +using System; +using System.Reflection; +using Vintagestory; +using Vintagestory.API.Config; +using Vintagestory.Client.NoObf; + +namespace Optimum.Render.Vulkan.Platform; + +/// +/// The client platform on the Vulkan path (Vulkan-native plan, Phase 1A). +/// +/// Created by in place of a +/// plain , so windowing, input, audio, the frame +/// pacing and the API-neutral render logic (post chain, TAA windows) are inherited. +/// It owns graphics bring-up and teardown and, since Phase 1A step 4, every graphics +/// operation: the partial files override each graphics member with calls to the +/// this platform created, and ClientPlatformWindows keeps +/// only the GL path. +/// +/// Compiled against the donor lib and bound at runtime to the Cecil-patched one, +/// so first checks that the loaded lib really +/// declares the virtuals this class relies on, and fails the install (OpenGL +/// fallback) instead of letting a call bypass an override mid-frame. +/// +public partial class VulkanClientPlatform : ClientPlatformWindows +{ + /// + /// The device this platform brought up in . Every + /// graphics override in the partial files calls it directly; null before a successful + /// install and after . + /// + private VulkanDevice device; + + /// Test seam: the device the overrides draw with. + internal VulkanDevice? GraphicsDevice => device; + + public const string ForceInstallFailureVariable = "OPTIMUM_VULKAN_FORCE_INSTALL_FAILURE"; + public const string ForcedInstallFailureReason = "forced by " + ForceInstallFailureVariable; + + /// A virtual the loaded lib must declare, matched by name and parameter type names. + internal readonly record struct ExpectedVirtual(bool OnAbstract, string Name, string[] ParameterTypeNames); + + /// + /// The injected virtuals on and the members + /// the patcher virtualizes in place on + /// (Optimum.Patcher/Program.cs, methodsToVirtualize). + /// + internal static readonly ExpectedVirtual[] ExpectedVirtuals = + { + new(true, "InitializeGraphics", new[] { "IntPtr", "Int32", "Int32", "String&" }), + new(true, "ShutdownGraphics", Array.Empty()), + new(false, "SetupDefaultFrameBuffers", Array.Empty()), + new(false, "DisposeFrameBuffers", new[] { "List`1" }), + new(false, "RenderFullscreenTriangle", new[] { "MeshRef" }), + new(false, "GetGraphicsCardRenderer", Array.Empty()), + new(false, "LogAndTestHardwareInfosStage2", Array.Empty()), + // Phase 1A step 3: program, uniform and UBO operations (overridden since step 4). + new(true, "UseShaderProgram", new[] { "Int32" }), + new(true, "DisposeShaderProgram", new[] { "ShaderProgramBase" }), + new(true, "BindSampler", new[] { "Int32", "Int32" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Single" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Int32" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Single", "Single" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Single", "Single", "Single" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Single", "Single", "Single", "Single" }), + new(true, "SetUniform", new[] { "Int32", "Int32", "Int32", "Int32", "Int32" }), + new(true, "SetUniformArray1", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "SetUniformArray2", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "SetUniformArray3", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "SetUniformArray4", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "SetUniformMatrix", new[] { "Int32", "Int32", "Single[]" }), + new(true, "SetUniformMatrix", new[] { "Int32", "Int32", "Matrix4&" }), + new(true, "SetUniformMatrices", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "SetUniformMatrices4x3", new[] { "Int32", "Int32", "Int32", "Single[]" }), + new(true, "BindProgramTexture2D", new[] { "ShaderProgramBase", "String", "Int32", "Int32" }), + new(true, "BindProgramTextureCube", new[] { "ShaderProgramBase", "String", "Int32", "Int32" }), + new(true, "BindUBO", new[] { "UBO" }), + new(true, "UnbindUBO", new[] { "UBO" }), + new(true, "UpdateUBO", new[] { "UBO", "IntPtr", "Int32", "Int32", "Boolean" }), + new(true, "DeleteUBO", new[] { "UBO" }), + // Phase 1A step 4: TAA motion windows and FSR target selection. + new(true, "EnableMotionDrawBuffers", Array.Empty()), + new(true, "RestorePrimaryDrawBuffers", Array.Empty()), + new(true, "EnableMotionOnlyDrawBuffers", Array.Empty()), + new(true, "ApplyOptimumMotionBlendState", Array.Empty()), + new(true, "ApplyOptimumMotionAccumulateBlendState", Array.Empty()), + new(true, "SelectFsrDrawBuffer", new[] { "FrameBufferRef" }), + // Phase 1A step 4: framebuffer binding, clears and post-chain pass state. + new(true, "BindCurrentFrameBuffer", new[] { "FrameBufferRef" }), + new(true, "BindCurrentFrameBufferKeepViewport", new[] { "FrameBufferRef" }), + new(true, "ClearBoundFrameBuffer", new[] { "FrameBufferRef", "Single[]", "Boolean", "Boolean" }), + new(true, "ClearFrameBufferPass", new[] { "EnumFrameBuffer" }), + new(true, "ApplyTransparentPassBlendState", Array.Empty()), + new(true, "SelectBackDrawBuffer", Array.Empty()), + new(true, "SetBlendEnabled", new[] { "Boolean" }), + new(true, "ApplyTransparentMergeBlendState", Array.Empty()), + new(true, "ClearSsaoTarget", Array.Empty()), + new(true, "BeginFinalCompositionDrawBuffers", Array.Empty()), + new(true, "RestoreWorldDrawBuffers", new[] { "Boolean" }), + // Phase 1A step 4: frame bracket, thick-line probe, window size, parity readback. + new(true, "BeginFrame", Array.Empty()), + new(true, "EndFrame", Array.Empty()), + new(true, "ProbeThickLineSupport", Array.Empty()), + new(true, "OnWindowSizeChanged", new[] { "Int32", "Int32" }), + new(true, "ReadTextureForParity", new[] { "Int32" }), + // Optimum AO: the platform's own ambient occlusion and its debug outputs. + new(true, "RenderOptimumAmbientOcclusion", new[] { "Single[]" }), + new(true, "OptimumAmbientOcclusionDebugTexture", new[] { "Int32" }), + // Phase 1A step 5: the leaf operations the render systems outside the platform issued. + new(true, "SetDepthRange", new[] { "Single", "Single" }), + new(true, "ClearDefaultDepth", new[] { "Single" }), + new(true, "DeleteMeshHandle", new[] { "Int32" }), + new(true, "DeleteVertexArrayHandles", new[] { "VAO" }), + new(true, "SetTextureLodBias", new[] { "Int32[]", "Single" }), + new(true, "SetSamplerLodBias", new[] { "Int32", "Single" }), + new(true, "SetTextureDepthCompare", new[] { "Int32", "Int32" }), + new(true, "ClearTextureRegion", new[] { "Int32", "Int32", "Int32", "Int32", "Int32", "Int32[]" }), + new(true, "LoadTextureFromRgbaPointer", new[] { "Int32", "Int32", "IntPtr" }), + new(true, "SetProgramSamplerUnit", new[] { "Int32", "String", "Int32" }), + new(true, "CreateOitTargets", new[] { "FrameBufferRef", "Int32", "Int32&", "Int32&" }), + new(true, "BeginOitAccumulation", new[] { "FrameBufferRef" }), + new(true, "BindOitTextures", new[] { "Int32", "Int32" }), + new(true, "GenOcclusionQuery", Array.Empty()), + new(true, "BeginOcclusionQuery", new[] { "Int32" }), + new(true, "EndOcclusionQuery", new[] { "Int32" }), + new(true, "TryGetOcclusionQueryResult", new[] { "Int32", "Int32&" }), + new(true, "DeleteOcclusionQuery", new[] { "Int32" }), + new(true, "ReadDefaultFramebuffer", new[] { "Int32", "Int32", "Int32", "Int32", "IntPtr" }), + new(true, "get_GraphicsBackendName", Array.Empty()), + // Headless render harness: the channel order ReadDefaultFramebuffer leaves + // behind. Not overridden here - the platform converts the device's R8G8B8A8 + // texels to the GL path's B G R A - but the virtual has to exist in the + // patched lib, because the harness reads it to decide how to write a frame. + new(true, "get_OptimumDefaultFramebufferIsBgra", Array.Empty()), + // Phase 2: render-stage bracket from ClientMain.TriggerRenderStage (contract C3). + new(true, "BeginRenderStage", new[] { "EnumRenderStage" }), + new(true, "EndRenderStage", new[] { "EnumRenderStage" }), + // "Latency seams" S3: the pre-input sleep and the frame-cap ownership flag. + new(true, "LatencySleep", Array.Empty()), + new(true, "get_LatencyOwnsFrameCap", Array.Empty()), + new(true, "SetLatencyFrameCap", new[] { "Int32" }), + // World/UI separation: the compose ClientMain and ScreenManager call. + new(true, "OptimumComposeUiTarget", Array.Empty()), + // Phase 2 step 2: the TAA post methods declare their frame-graph passes. + new(true, "RenderOptimumSkyMotion", Array.Empty()), + // Phase 3b stage 2: the sky dome's draw seam, the first world system on the native API. + new(true, "RenderSkyDome", new[] { "MeshRef", "Int32", "Int32", "Single[]" }), + // Phase 3b stage 2: the chunk draw-group scope every ChunkRenderer pass brackets + // its pools with, which routes the terrain multi-draws through the native API. + new(true, "BeginChunkPass", new[] { "String", "Boolean", "Boolean", "Boolean", "Boolean" }), + new(true, "EndChunkPass", Array.Empty()), + // Phase 3b stage 2: the entity draw seam - every sub-mesh of a multi-texture mesh. + new(true, "RenderEntityMesh", new[] { "MeshRef", "String", "Int32" }), + // Phase 3b stage 2: the remaining sky, particle and decal draw seams. + new(true, "RenderNightSkyBox", new[] { "MeshRef", "Int32" }), + new(true, "RenderCelestialQuad", new[] { "MeshRef", "Int32", "Int32", "Int32" }), + new(true, "RenderSunQuad", new[] { "MeshRef", "Int32" }), + new(true, "RenderGuiQuad", new[] { "MeshRef", "Int32" }), + new(true, "RenderParticles", new[] { "MeshRef", "Int32", "Int32" }), + new(true, "BeginDecalPass", new[] { "Int32", "Int32" }), + new(true, "EndDecalPass", Array.Empty()), + // Phase 3b stage 2, GUI and text: the texture-into-texture blit and the aiming reticle's + // line draws, the two GUI systems whose fixed state is stated at their call site. + new(true, "RenderTextureQuad", new[] { "MeshRef", "Int32", "Boolean" }), + new(true, "RenderOverlayLines", new[] { "MeshRef", "Int32", "Single", "Boolean" }), + new(true, "RenderOptimumTaaResolve", Array.Empty()), + new(true, "RenderOptimumTaaSharpen", new[] { "Int32" }), + // Phase 3b stage 1: the two TAA passes' draw seams, which the native chain replaces. + new(true, "OptimumTaaResolveDraw", + new[] { "FrameBufferRef", "FrameBufferRef", "Single[]", "Single[]", "Boolean" }), + new(true, "OptimumTaaSharpenDraw", new[] { "FrameBufferRef", "Int32" }), + }; + + /// + /// Non-virtual members injected into that the + /// overrides read or call (the platform state behind the device framebuffer setup and + /// the SSAO flag). A lib without them would fail with MissingMethodException mid-frame. + /// + internal static readonly string[] ExpectedWindowsMembers = + { + "OptimumRenderSsao", + "OptimumAdoptFrameBufferSettings", + "OptimumTaaRequested", + "OptimumSsaoKernel", + "SetOptimumMotionAttachmentIndex", + "OptimumAdoptTaaTargets", + "OptimumFinishDeviceFrameBufferSetup", + }; + + /// + /// Test seam: the device to bring up (tests add validation capture). The real one keeps + /// compiled shaders and the pipeline cache in the game's per-user cache folder. + /// + internal Func DeviceFactory = () => new VulkanDevice + { + ShaderCacheDirectory = System.IO.Path.Combine(GamePaths.Cache, "optimum-vulkan"), + ShaderProgramOverriddenByMods = Vintagestory.API.Config.OptimumConfig.IsShaderProgramOverriddenByMods, + }; + + /// Test seam: where the crash marker goes; null means . + internal string? CrashMarkerDataPath; + + public VulkanClientPlatform(Logger logger) : base(logger) + { + } + + /// + /// Checks the loaded lib against . Reflection + /// only; never throws. + /// + internal static bool VerifyHost(Type abstractType, Type windowsType, out string? reason) + { + try + { + if (windowsType.IsSealed) + { + reason = windowsType.FullName + " is sealed in the loaded VintagestoryLib (not patched for this renderer)"; + return false; + } + if (!windowsType.IsSubclassOf(abstractType)) + { + reason = windowsType.FullName + " does not derive from " + abstractType.FullName; + return false; + } + + foreach (ExpectedVirtual expected in ExpectedVirtuals) + { + Type owner = expected.OnAbstract ? abstractType : windowsType; + MethodInfo? method = FindDeclared(owner, expected); + if (method == null || !method.IsVirtual || method.IsFinal) + { + reason = "the loaded VintagestoryLib lacks the virtual " + owner.Name + "." + expected.Name + + " (not patched for this renderer)"; + return false; + } + } + + const BindingFlags memberFlags = BindingFlags.Public | BindingFlags.Instance | BindingFlags.DeclaredOnly; + foreach (string name in ExpectedWindowsMembers) + { + if (windowsType.GetMember(name, memberFlags).Length == 0) + { + reason = "the loaded VintagestoryLib lacks " + windowsType.Name + "." + name + + " (not patched for this renderer)"; + return false; + } + } + + reason = null; + return true; + } + catch (Exception error) + { + reason = "the platform self-check threw: " + error.Message; + return false; + } + } + + private static MethodInfo? FindDeclared(Type owner, ExpectedVirtual expected) + { + const BindingFlags flags = BindingFlags.Public | BindingFlags.Instance | BindingFlags.DeclaredOnly; + foreach (MethodInfo method in owner.GetMethods(flags)) + { + if (method.Name != expected.Name) continue; + ParameterInfo[] parameters = method.GetParameters(); + if (parameters.Length != expected.ParameterTypeNames.Length) continue; + bool matches = true; + for (int i = 0; i < parameters.Length && matches; i++) + matches = parameters[i].ParameterType.Name == expected.ParameterTypeNames[i]; + if (matches) return method; + } + return null; + } + + internal static bool IsInstallFailureForced() => + Environment.GetEnvironmentVariable(ForceInstallFailureVariable) == "1"; + + /// + /// Creates the Vulkan device for the window, marks the backend Vulkan and publishes + /// the fork graphics bridge. False leaves the caller holding a window with no graphics + /// API, which it reopens for OpenGL with a base platform. + /// + public override bool InitializeGraphics(IntPtr windowHandle, int width, int height, out string reason) + { + string? hostReason; + if (!VerifyHost(typeof(ClientPlatformAbstract), typeof(ClientPlatformWindows), out hostReason)) + { + reason = hostReason!; + return false; + } + + if (IsInstallFailureForced()) + { + reason = ForcedInstallFailureReason; + return false; + } + + reason = null!; + + // Installing is a single transition: a device this platform already brought + // up stays, so a second call cannot displace and leak the one the client is + // drawing with. + if (this.device != null) + { + return true; + } + + VulkanDevice? device = null; + try + { + device = DeviceFactory(); + + // The marker goes down before the driver is touched: a crash inside + // device creation is exactly the kind the next start must see. A + // clean failure clears it again, since the caller falls back to + // OpenGL on its own. + OptimumRenderBootstrap.WriteCrashMarker(CrashMarkerDataPath ?? GamePaths.DataPath); + + if (!device.Initialize(windowHandle, width, height, out string failureReason)) + { + device.Dispose(); + OptimumRenderBootstrap.ClearCrashMarker(); + reason = failureReason; + return false; + } + + this.device = device; + device.OwnerPlatform = this; + // Phase 2 step 2: the stage bracket drives the frame graph's pass declarations. + RenderStageListener = new FrameGraphStageListener(this); + // Phase 5: registered mod motion writers reach this platform's motion window. + InstallModPassHooks(); + OptimumRender.ActiveBackend = EnumRenderBackend.Vulkan; + OptimumForkGraphics.Active = new VulkanForkGraphics(this, device); + return true; + } + catch (Exception error) + { + try + { + device?.Dispose(); + } + catch (Exception) + { + // The install already failed; the reason below is the useful one. + } + OptimumRenderBootstrap.ClearCrashMarker(); + reason = error.Message; + return false; + } + } + + /// + /// Shuts the device down, returns the backend state to OpenGL and clears the + /// crash marker. Safe to call more than once and after a failed install. + /// + public override void ShutdownGraphics() + { + // The bridge goes first: nothing may reach a device that is being torn down. + OptimumForkGraphics.Active = null; + RemoveModPassHooks(); + try + { + ReleaseAmbientOcclusion(); + } + catch (Exception) + { + // Released with the device below either way. + } + try + { + device?.Dispose(); + } + catch (Exception) + { + // A driver throwing on teardown must not stop the client exiting. + } + + device = null; + RenderStageListener = null; + OptimumRender.ActiveBackend = EnumRenderBackend.OpenGL; + OptimumRender.NoGraphicsApiWindow = false; + OptimumRenderBootstrap.ClearCrashMarker(); + } +} diff --git a/Optimum.Render.Vulkan/Platform/VulkanForkGraphics.cs b/Optimum.Render.Vulkan/Platform/VulkanForkGraphics.cs new file mode 100644 index 00000000..763bc822 --- /dev/null +++ b/Optimum.Render.Vulkan/Platform/VulkanForkGraphics.cs @@ -0,0 +1,98 @@ +using System; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +namespace Optimum.Render.Vulkan.Platform; + +/// +/// The the Vulkan platform publishes while its graphics +/// are up: the handful of operations the forked VSEssentials and VSSurvivalMod renderers +/// need beyond IRenderAPI, forwarded unchanged to the platform's device. The forks +/// reference only the API and the contracts, so this is how they reach the device until +/// Phase 5 ports them. +/// +/// The state operations (binds, viewport, draw buffers, depth test, blend, texture units) are +/// recorded on the platform only, as the platform's own state virtuals are +/// (VulkanClientPlatform.State.cs): the native draws of the fork's RenderMesh read the state +/// the fork set and the target it bound from there. +/// +internal sealed class VulkanForkGraphics : OptimumForkGraphics +{ + private readonly VulkanDevice device; + private readonly VulkanClientPlatform platform; + + public VulkanForkGraphics(VulkanClientPlatform platform, VulkanDevice device) + { + this.platform = platform; + this.device = device; + } + + public override int CreateTexture2DRaw(int width, int height, int glInternalFormat, IntPtr pixels, int bytesPerPixel) => + device.CreateTexture2DRaw(width, height, glInternalFormat, pixels, bytesPerPixel); + + public override int CreateTexture2DArray(int width, int height, int layers, + EnumTextureInternalFormat internalFormat, EnumTexturePixelFormat pixelFormat) => + device.CreateTexture2DArray(width, height, layers, internalFormat, pixelFormat); + + public override void UploadTexture2DArrayLayer(int textureId, int layer, int x, int y, int width, int height, IntPtr pixels) => + device.UploadTexture2DArrayLayer(textureId, layer, x, y, width, height, pixels); + + public override void UploadTexture2DNormalizedShorts(int textureId, int level, int x, int y, int width, int height, short[] pixels) => + device.UploadTexture2DNormalizedShorts(textureId, level, x, y, width, height, pixels); + + public override void SetTextureParameter(int textureId, int parameterName, int value) => + device.SetTextureParameter(textureId, parameterName, value); + + public override void BindTexture(int unit, int textureId) + { + platform.NoteForkTexture(unit, textureId); + } + + public override void DeleteTexture(int textureId) => device.DeleteTexture(textureId); + + public override int CreateFramebuffer(int width, int height) => device.CreateFramebuffer(width, height); + + public override void AttachTexture(int framebufferId, EnumFramebufferAttachment attachment, int textureId, int layer) => + device.AttachTexture(framebufferId, attachment, textureId, layer); + + public override void SetDrawBuffers(int framebufferId, int attachmentMask) + { + platform.NoteForkDrawBuffers(framebufferId, attachmentMask); + } + + public override void BindFramebuffer(int framebufferId) + { + platform.NoteForkFramebuffer(framebufferId); + } + + public override void BindDefaultFramebuffer() + { + platform.NoteForkFramebuffer(0); + } + + public override void DeleteFramebuffer(int framebufferId) + { + platform.stated.ForgetFramebuffer(framebufferId); + device.DeleteFramebuffer(framebufferId); + } + + public override void SetViewport(int x, int y, int width, int height) + { + platform.NoteForkViewport(x, y, width, height); + } + + public override void SetDepthTest(bool enabled) + { + platform.NoteForkDepthTest(enabled); + } + + public override void SetBlendEnabled(bool enabled) + { + platform.NoteForkBlend(enabled); + } + + public override int GetUniformLocation(int programId, string name) => device.GetUniformLocation(programId, name); + + public override void SetUniformArray3(int programId, int location, int count, float[] values) => + device.SetUniformArray3(programId, location, count, values); +} diff --git a/Optimum.Render.Vulkan/Present/Swapchain.cs b/Optimum.Render.Vulkan/Present/Swapchain.cs new file mode 100644 index 00000000..f19877e9 --- /dev/null +++ b/Optimum.Render.Vulkan/Present/Swapchain.cs @@ -0,0 +1,872 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Graph; +using Silk.NET.Vulkan; +using Silk.NET.Vulkan.Extensions.KHR; + +using Semaphore = Silk.NET.Vulkan.Semaphore; + +namespace Optimum.Render.Vulkan.Core; + +/// +/// The process-wide present id: one +/// value per vkQueuePresentKHR, monotonically increasing and never reset. +/// +/// It is deliberately not the latency frame id and not a Frame timeline value. +/// The Frame timeline advances two or three times per frame, and the frame id is +/// allocated once per rendered frame; the present id counts presents, which is +/// what VK_KHR_present_id and every vendor's present-timing query mean by +/// it. One frame maps to one present today, and the map below keeps that pairing +/// explicit across partial submissions and swapchain recreation. +/// +/// Global rather than per swapchain so the sequence survives recreation: a +/// resize, a vsync toggle or an OUT_OF_DATE rebuild must not restart it. +/// +internal static class PresentIdCounter +{ + private static long _next; + + /// The next present id; the first is 1. + public static ulong Next() => (ulong)System.Threading.Interlocked.Increment(ref _next); + + /// The last id handed out, 0 before the first present. + public static ulong Current => (ulong)System.Threading.Interlocked.Read(ref _next); +} + +/// +/// The last presents' frame id per present id, kept small and wrapping: enough +/// to answer "which frame was present id N" for the frames a driver report can +/// still be about, never a growing map. +/// +internal sealed class PresentIdMap +{ + public const int DefaultCapacity = 64; + + private readonly ulong[] _presentIds; + private readonly ulong[] _frameIds; + private int _next; + + public PresentIdMap(int capacity = DefaultCapacity) + { + _presentIds = new ulong[capacity]; + _frameIds = new ulong[capacity]; + } + + /// The newest present id recorded, 0 before the first. + public ulong LastPresentId { get; private set; } + + /// The frame id of the newest present recorded, 0 before the first. + public ulong LastFrameId { get; private set; } + + public void Record(ulong presentId, ulong frameId) + { + _presentIds[_next] = presentId; + _frameIds[_next] = frameId; + _next = (_next + 1) % _presentIds.Length; + LastPresentId = presentId; + LastFrameId = frameId; + } + + /// The frame that produced , while it is still remembered. + public bool TryGetFrameId(ulong presentId, out ulong frameId) + { + for (int i = 0; i < _presentIds.Length; i++) + { + if (_presentIds[i] == presentId && presentId != 0) + { + frameId = _frameIds[i]; + return true; + } + } + frameId = 0; + return false; + } +} + +/// An acquired swapchain image and the semaphores its present submission uses. +internal readonly struct PresentTarget +{ + public PresentTarget(SwapchainSlot slot, uint imageIndex, Semaphore acquireSemaphore) + { + Slot = slot; + ImageIndex = imageIndex; + AcquireSemaphore = acquireSemaphore; + } + + public SwapchainSlot Slot { get; } + public uint ImageIndex { get; } + + /// Signalled by the acquire; Submit B waits on it. + public Semaphore AcquireSemaphore { get; } + + /// Signalled by Submit B; vkQueuePresentKHR waits on it. + public Semaphore PresentSemaphore => Slot.PresentSemaphoreFor(ImageIndex); + + public Image Image => Slot.Images[ImageIndex]; + public Extent2D Extent => Slot.Extent; +} + +/// +/// One vkCreateSwapchainKHR result and everything that belongs to it: images, +/// views, the acquire-semaphore free list (imageCount + 1) and one present +/// semaphore per image. Created by and retired as one +/// unit through once a successor image +/// reacquisition proves presentation has released it, so its semaphores die with it. +/// +internal sealed unsafe class SwapchainSlot : IDisposable +{ + private readonly VulkanContext _context; + private readonly KhrSwapchain _api; + private readonly Semaphore[] _acquireSemaphores; + private readonly Semaphore[] _presentSemaphores; + private readonly AcquireSemaphoreFreeList _freeAcquire; + private readonly List _pendingPresents = new(); + private readonly Stack _freePresentFences = new(); + private bool _disposed; + + public SwapchainSlot(VulkanContext context, KhrSwapchain api, SwapchainKHR handle, + Extent2D extent, Format format, PresentModeKHR presentMode) + { + _context = context; + _api = api; + Handle = handle; + Extent = extent; + Format = format; + PresentMode = presentMode; + + uint count = 0; + api.GetSwapchainImages(context.Device, handle, ref count, null); + Images = new Image[count]; + fixed (Image* imagesPtr = Images) + { + api.GetSwapchainImages(context.Device, handle, ref count, imagesPtr); + } + + Views = new ImageView[count]; + for (int i = 0; i < count; i++) + { + var viewInfo = new ImageViewCreateInfo + { + SType = StructureType.ImageViewCreateInfo, + Image = Images[i], + ViewType = ImageViewType.Type2D, + Format = format, + SubresourceRange = new ImageSubresourceRange(ImageAspectFlags.ColorBit, 0, 1, 0, 1), + }; + ImageView view; + VulkanResult.Check(context.Api.CreateImageView(context.Device, &viewInfo, null, &view), + "vkCreateImageView for a swapchain image"); + Views[i] = view; + } + + // The present semaphore belongs to the IMAGE, not to a rolling counter: + // vkQueuePresentKHR keeps waiting on it until that image is presented, and + // the only moment it is provably free again is when the same image is + // re-acquired. A counter-indexed semaphore could be re-signalled while an + // earlier present still waits on it. + _presentSemaphores = CreateSemaphores((int)count); + _acquireSemaphores = CreateSemaphores(AcquireSemaphoreFreeList.CapacityFor(count)); + var handles = new ulong[_acquireSemaphores.Length]; + for (int i = 0; i < handles.Length; i++) handles[i] = _acquireSemaphores[i].Handle; + _freeAcquire = new AcquireSemaphoreFreeList(handles); + } + + public SwapchainKHR Handle { get; } + public Image[] Images { get; } + public ImageView[] Views { get; } + public Extent2D Extent { get; } + public Format Format { get; } + public PresentModeKHR PresentMode { get; } + public uint ImageCount => (uint)Images.Length; + public int AcquireSemaphoreCount => _acquireSemaphores.Length; + public int FreeAcquireSemaphores => _freeAcquire.FreeCount; + + /// Frame timeline value of the newest present submission that used one of this slot's images; 0 before the first. + public ulong LastPresentValue { get; private set; } + + public uint? FirstPresentedImage { get; private set; } + + public void NotePresented(uint imageIndex) => FirstPresentedImage ??= imageIndex; + + public Semaphore PresentSemaphoreFor(uint imageIndex) => _presentSemaphores[imageIndex]; + + public int PendingAcquireSemaphores => _freeAcquire.PendingCount; + + /// A semaphore for the next acquire; releases those whose present submission finished. + public Semaphore TakeAcquireSemaphore(ulong frameCompleted) => new(_freeAcquire.Take(frameCompleted)); + + /// The acquire failed; the semaphore is untouched. + public void ReturnAcquireSemaphore(Semaphore semaphore) => _freeAcquire.Return(semaphore.Handle); + + /// A submission carrying waits on the semaphore; reusable once it completed. + public void ReturnAcquireSemaphoreAfter(Semaphore semaphore, ulong frameValue) => + _freeAcquire.ReturnAfter(semaphore.Handle, frameValue); + + public void NotePresentSubmitted(ulong frameValue) + { + if (frameValue > LastPresentValue) LastPresentValue = frameValue; + } + + public bool PresentsComplete() + { + for (int i = _pendingPresents.Count - 1; i >= 0; i--) + { + Fence fence = _pendingPresents[i]; + Result status = _context.Api.GetFenceStatus(_context.Device, fence); + if (status == Result.NotReady) continue; + VulkanResult.Check(status, "vkGetFenceStatus for presentation"); + _pendingPresents.RemoveAt(i); + _freePresentFences.Push(fence); + } + return _pendingPresents.Count == 0; + } + + public Fence PreparePresentFence() + { + if (!_context.Capabilities.PresentFencesEnabled) return default; + PresentsComplete(); + if (_freePresentFences.TryPop(out Fence fence)) + VulkanResult.Check(_context.Api.ResetFences(_context.Device, 1, &fence), "vkResetFences for presentation"); + else + { + var info = new FenceCreateInfo { SType = StructureType.FenceCreateInfo }; + VulkanResult.Check(_context.Api.CreateFence(_context.Device, &info, null, &fence), "vkCreateFence for presentation"); + } + return fence; + } + + public void FinishPresentFence(Fence fence, Result result) + { + if (fence.Handle == 0) return; + // Memory failures leave synchronization primitives untouched. Other present + // errors either enqueue the operation or lose the device. + if (result is Result.ErrorOutOfHostMemory or Result.ErrorOutOfDeviceMemory) + _freePresentFences.Push(fence); + else + _pendingPresents.Add(fence); + } + + private Semaphore[] CreateSemaphores(int count) + { + var semaphores = new Semaphore[count]; + var createInfo = new SemaphoreCreateInfo { SType = StructureType.SemaphoreCreateInfo }; + for (int i = 0; i < count; i++) + { + Semaphore semaphore; + VulkanResult.Check(_context.Api.CreateSemaphore(_context.Device, &createInfo, null, &semaphore), + "vkCreateSemaphore for a swapchain slot"); + semaphores[i] = semaphore; + } + return semaphores; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + // Teardown waits; ordinary retirement polls these fences before disposing. + foreach (Fence pending in _pendingPresents) + { + Fence fence = pending; + Result result = _context.Api.WaitForFences(_context.Device, 1, &fence, true, ulong.MaxValue); + if (result != Result.ErrorDeviceLost) VulkanResult.Check(result, "vkWaitForFences for presentation teardown"); + _context.Api.DestroyFence(_context.Device, fence, null); + } + foreach (Fence fence in _freePresentFences) _context.Api.DestroyFence(_context.Device, fence, null); + _pendingPresents.Clear(); + _freePresentFences.Clear(); + + foreach (ImageView view in Views) + { + if (view.Handle != 0) _context.Api.DestroyImageView(_context.Device, view, null); + } + if (Handle.Handle != 0) _api.DestroySwapchain(_context.Device, Handle, null); + foreach (Semaphore semaphore in _acquireSemaphores) + { + if (semaphore.Handle != 0) _context.Api.DestroySemaphore(_context.Device, semaphore, null); + } + foreach (Semaphore semaphore in _presentSemaphores) + { + if (semaphore.Handle != 0) _context.Api.DestroySemaphore(_context.Device, semaphore, null); + } + } +} + +/// +/// The presentation chain. +/// +/// Recreation follows the Khronos swapchain_recreation sample: the current +/// swapchain is always passed as oldSwapchain, nothing waits for the +/// device to go idle, and the replaced is retired on +/// the Frame timeline after the last present submission that used it. SUBOPTIMAL +/// (from acquire or present) rebuilds before the next acquire; OUT_OF_DATE +/// rebuilds and acquires once more; a zero extent (a minimised window) parks +/// presentation until the surface grows again. +/// +/// The one flip of the image happens in the present path +/// (), not here. +/// +internal sealed unsafe class Swapchain : IDisposable +{ + /// OPTIMUM_VULKAN_FIFO_RELAXED=0 forces plain FIFO (no promotion on missed vsyncs). + public const string FifoRelaxedVariable = "OPTIMUM_VULKAN_FIFO_RELAXED"; + + private readonly VulkanContext _context; + private readonly KhrSurface _surfaceApi; + private readonly KhrSwapchain _swapchainApi; + private readonly SurfaceKHR _surface; + private readonly SwapchainRetirement _retirement; + private readonly ITimelineClock _clock; + private readonly bool _relaxedAllowed; + + private SwapchainSlot? _current; + private uint _width; + private uint _height; + private bool _vsync; + private bool _relaxedPromoted; + private bool _disposed; + + public Format Format { get; private set; } = Format.B8G8R8A8Unorm; + public Extent2D Extent { get; private set; } + public PresentModeKHR PresentMode { get; private set; } = PresentModeKHR.FifoKhr; + public uint ImageCount => _current?.ImageCount ?? 0; + + /// Set when the chain is stale; the next acquire rebuilds it first. + public bool NeedsRecreation { get; private set; } + + /// The surface has zero extent; nothing is acquired or presented until it grows. + public bool Parked { get; private set; } + + /// Swapchains created so far, the first included. + public int Creations { get; private set; } + + /// Replaced slots still waiting for the GPU. + public int RetiredPending => _retirement.PendingCount; + + /// Why the last rebuild failed; null after a successful one. + public string? RebuildFailure { get; private set; } + + /// The slot being acquired from. Tests only. + internal SwapchainSlot? CurrentSlotForTests => _current; + + /// + /// Whether VkPresentIdKHR may be chained onto the present (seam S2). + /// Set from the capabilities when VK_KHR_present_id is enabled AND its feature + /// was turned on; chaining it otherwise is a validation error, so it stays off + /// by default. The id itself is allocated either way, so the frame to present + /// mapping does not depend on the extension. + /// + internal bool PresentIdEnabled { get; set; } + + /// The frame id of each of the last presents, by present id (seam S2). + internal PresentIdMap PresentIds { get; } = new(); + + private Swapchain(VulkanContext context, KhrSurface surfaceApi, KhrSwapchain swapchainApi, SurfaceKHR surface, + ITimelineClock clock) + { + _context = context; + _surfaceApi = surfaceApi; + _swapchainApi = swapchainApi; + _surface = surface; + _clock = clock; + _retirement = new SwapchainRetirement(clock); + _relaxedAllowed = Environment.GetEnvironmentVariable(FifoRelaxedVariable)?.Trim() != "0"; + } + + /// + /// Takes ownership of on entry: on every failure + /// return the surface is destroyed here, and on success the swapchain + /// destroys it in . The caller never destroys it. + /// + public static bool TryCreate( + VulkanContext context, SurfaceKHR surface, uint width, uint height, bool vsync, ITimelineClock clock, + out Swapchain? swapchain, out string? failureReason) + { + swapchain = null; + failureReason = null; + + if (!context.Api.TryGetInstanceExtension(context.Instance, out KhrSurface surfaceApi)) + { + failureReason = "VK_KHR_surface unavailable"; + WindowSurface.Destroy(context, surface); + return false; + } + if (!context.Api.TryGetDeviceExtension(context.Instance, context.Device, out KhrSwapchain swapchainApi)) + { + failureReason = "VK_KHR_swapchain unavailable"; + surfaceApi.DestroySurface(context.Instance, surface, null); + surfaceApi.Dispose(); + return false; + } + + // The graphics queue has to be able to present. A separate present queue + // is possible in principle but does not occur on any desktop driver, and + // supporting it would add a queue-ownership transfer to every frame. + surfaceApi.GetPhysicalDeviceSurfaceSupport( + context.PhysicalDevice, context.GraphicsQueueFamily, surface, + out Silk.NET.Core.Bool32 supported); + if (!supported) + { + failureReason = "the graphics queue family cannot present to this surface"; + surfaceApi.DestroySurface(context.Instance, surface, null); + surfaceApi.Dispose(); + swapchainApi.Dispose(); + return false; + } + + var created = new Swapchain(context, surfaceApi, swapchainApi, surface, clock); + created._width = width; + created._height = height; + created._vsync = vsync; + if (!created.Build(out failureReason)) + { + // A window that starts minimised is not a device the client can use. + if (created.Parked) failureReason = "surface has zero extent"; + created.Dispose(); + return false; + } + + swapchain = created; + return true; + } + + /// + /// Builds a new slot from the current surface state, passing the current one + /// as oldSwapchain and retiring it. Never waits. Returns false when parked (the + /// current slot is kept for the rebuild that unparks) or when creation failed. + /// + private bool Build(out string? failureReason) + { + failureReason = null; + + _surfaceApi.GetPhysicalDeviceSurfaceCapabilities( + _context.PhysicalDevice, _surface, out SurfaceCapabilitiesKHR capabilities); + + Extent2D extent = ChooseExtent(capabilities, _width, _height); + if (SwapchainPolicy.IsParked(extent)) + { + Parked = true; + NeedsRecreation = true; + return false; + } + Parked = false; + + Format = ChooseFormat(out ColorSpaceKHR colorSpace); + PresentModeKHR presentMode = SwapchainPolicy.ChoosePresentMode( + _vsync, _relaxedPromoted && _relaxedAllowed, SupportedPresentModes()); + uint imageCount = SwapchainPolicy.ChooseImageCount( + capabilities.MinImageCount, capabilities.MaxImageCount, presentMode); + + SwapchainSlot? old = _current; + var createInfo = new SwapchainCreateInfoKHR + { + SType = StructureType.SwapchainCreateInfoKhr, + Surface = _surface, + MinImageCount = imageCount, + ImageFormat = Format, + ImageColorSpace = colorSpace, + ImageExtent = extent, + ImageArrayLayers = 1, + // Transfer destination because the frame is blitted in rather than + // rendered directly: the game renders into its own targets and the + // last step copies the result across, flipping it on the way. + ImageUsage = ImageUsageFlags.ColorAttachmentBit | ImageUsageFlags.TransferDstBit, + ImageSharingMode = SharingMode.Exclusive, + PreTransform = capabilities.CurrentTransform, + CompositeAlpha = CompositeAlphaFlagsKHR.OpaqueBitKhr, + PresentMode = presentMode, + Clipped = true, + // Always the current chain: images it has not handed out can be freed + // by the driver right away, and presents already queued on it finish. + OldSwapchain = old?.Handle ?? default, + }; + + Result result = _swapchainApi.CreateSwapchain(_context.Device, &createInfo, null, out SwapchainKHR handle); + + // Passing oldSwapchain retires it even when creation fails. + if (old != null) + { + _retirement.Retire(old, old.LastPresentValue, + _context.Capabilities.PresentFencesEnabled ? old.PresentsComplete : null); + _current = null; + } + + if (result != Result.Success) + { + failureReason = "vkCreateSwapchainKHR failed with " + result; + RebuildFailure = failureReason; + NeedsRecreation = true; + return false; + } + + _current = new SwapchainSlot(_context, _swapchainApi, handle, extent, Format, presentMode); + Extent = extent; + PresentMode = presentMode; + Creations++; + NeedsRecreation = false; + RebuildFailure = null; + return true; + } + + private Extent2D ChooseExtent(SurfaceCapabilitiesKHR capabilities, uint width, uint height) + { + // A driver that pins the extent wins; otherwise clamp what we asked for. + if (capabilities.CurrentExtent.Width != uint.MaxValue) + { + return capabilities.CurrentExtent; + } + + return new Extent2D( + Math.Clamp(width, capabilities.MinImageExtent.Width, capabilities.MaxImageExtent.Width), + Math.Clamp(height, capabilities.MinImageExtent.Height, capabilities.MaxImageExtent.Height)); + } + + /// + /// Prefers a plain 8-bit BGRA format in sRGB colour space. The game's default + /// framebuffer is linear - it never enables GL_FRAMEBUFFER_SRGB - so an + /// _SRGB image format would apply a conversion the GL path never did and + /// wash the picture out. + /// + private Format ChooseFormat(out ColorSpaceKHR colorSpace) + { + uint count = 0; + _surfaceApi.GetPhysicalDeviceSurfaceFormats(_context.PhysicalDevice, _surface, ref count, null); + + var formats = new SurfaceFormatKHR[count]; + fixed (SurfaceFormatKHR* formatsPtr = formats) + { + _surfaceApi.GetPhysicalDeviceSurfaceFormats(_context.PhysicalDevice, _surface, ref count, formatsPtr); + } + + foreach (SurfaceFormatKHR candidate in formats) + { + if (candidate.Format is Format.B8G8R8A8Unorm or Format.R8G8B8A8Unorm) + { + colorSpace = candidate.ColorSpace; + return candidate.Format; + } + } + + if (formats.Length > 0) + { + colorSpace = formats[0].ColorSpace; + return formats[0].Format; + } + + colorSpace = ColorSpaceKHR.SpaceSrgbNonlinearKhr; + return Format.B8G8R8A8Unorm; + } + + private PresentModeKHR[] SupportedPresentModes() + { + uint count = 0; + _surfaceApi.GetPhysicalDeviceSurfacePresentModes(_context.PhysicalDevice, _surface, ref count, null); + + var modes = new PresentModeKHR[count]; + fixed (PresentModeKHR* modesPtr = modes) + { + _surfaceApi.GetPhysicalDeviceSurfacePresentModes(_context.PhysicalDevice, _surface, ref count, modesPtr); + } + return modes; + } + + /// + /// Asks for a rebuild at the next acquire (a resize, a vsync toggle). Cheap + /// and repeatable: a resize storm costs one rebuild per presented frame at most. + /// + public void RequestRebuild(uint width, uint height, bool vsync) + { + if (vsync != _vsync) _relaxedPromoted = false; + _width = width; + _height = height; + _vsync = vsync; + NeedsRecreation = true; + } + + /// + /// Sustained missed vsyncs under FIFO: rebuild as FIFO_RELAXED when the + /// surface has it and the override allows it. Returns whether a rebuild was requested. + /// + public bool PromoteToRelaxedFifo() + { + if (!_vsync || _relaxedPromoted || !_relaxedAllowed) return false; + if (Array.IndexOf(SupportedPresentModes(), PresentModeKHR.FifoRelaxedKhr) < 0) return false; + _relaxedPromoted = true; + NeedsRecreation = true; + return true; + } + + /// + /// Acquires the next image, rebuilding first when the chain is stale. + /// Returns false when nothing can be presented this frame: parked, the chain + /// was still out of date after one rebuild, or the rebuild failed + /// (). A resize or a monitor change is ordinary, + /// never an error; a lost device throws. + /// + public bool TryAcquire(out PresentTarget target) + { + target = default; + _retirement.Collect(); + + if (NeedsRecreation || _current == null) + { + if (!Build(out _)) return false; + } + + for (int attempt = 0; attempt < 2; attempt++) + { + SwapchainSlot slot = _current!; + Semaphore acquire = slot.TakeAcquireSemaphore( + slot.FreeAcquireSemaphores == 0 ? _clock.FrameCompleted : 0); + uint imageIndex = 0; + + long waitStart = VulkanStats.WaitStart(); + if (_context.AcquireDelayForTests > TimeSpan.Zero) System.Threading.Thread.Sleep(_context.AcquireDelayForTests); + Result result = _swapchainApi.AcquireNextImage( + _context.Device, slot.Handle, ulong.MaxValue, acquire, default, ref imageIndex); + VulkanStats.NoteWait(WaitSite.SwapchainAcquire, waitStart); + + switch (SwapchainPolicy.OnAcquire(result, attempt)) + { + case AcquireAction.Present: + target = new PresentTarget(slot, imageIndex, acquire); + return true; + + case AcquireAction.PresentThenRebuild: + // Usable this frame; rebuilt before the next acquire. + NeedsRecreation = true; + target = new PresentTarget(slot, imageIndex, acquire); + return true; + + case AcquireAction.RebuildAndRetry: + slot.ReturnAcquireSemaphore(acquire); + NeedsRecreation = true; + if (!Build(out _)) return false; + continue; + + case AcquireAction.SkipFrame: + slot.ReturnAcquireSemaphore(acquire); + NeedsRecreation = true; + return false; + + default: + slot.ReturnAcquireSemaphore(acquire); + // Anything else - a lost device above all - is reported rather + // than turned into a quiet "no image this frame", which reads + // as a freeze. + VulkanResult.Check(result, "vkAcquireNextImageKHR"); + NeedsRecreation = true; + return false; + } + } + return false; + } + + /// + /// The present submission waiting on 's acquire + /// semaphore was accepted with Frame value : the + /// acquire semaphore is reusable once that value completes. If this reacquired + /// the successor's first presented image, completion also releases retired chains. + /// + public void NotePresentSubmitted(in PresentTarget target, ulong frameValue) + { + if (target.Slot.FirstPresentedImage == target.ImageIndex) + _retirement.NoteSuccessorReacquired(frameValue); + target.Slot.ReturnAcquireSemaphoreAfter(target.AcquireSemaphore, frameValue); + target.Slot.NotePresentSubmitted(frameValue); + } + + /// + /// Presents and returns the present id this present + /// was given (seam S2): one per call, increasing across swapchain recreation, + /// chained as VkPresentIdKHR when . + /// is the latency frame id that produced it, kept + /// in . + /// + public ulong Present(in PresentTarget target, ulong frameId = 0) + { + SwapchainKHR handle = target.Slot.Handle; + Semaphore wait = target.PresentSemaphore; + uint index = target.ImageIndex; + + // Allocated for every present, whether or not the extension carries it, + // so the frame to present mapping is the same on every driver. + ulong presentId = PresentIdCounter.Next(); + PresentIds.Record(presentId, frameId); + + var presentIdInfo = new PresentIdKHR + { + SType = StructureType.PresentIDKhr, + SwapchainCount = 1, + PPresentIds = &presentId, + }; + + Fence presentFence = target.Slot.PreparePresentFence(); + var fenceInfo = new SwapchainPresentFenceInfoEXT + { + SType = StructureType.SwapchainPresentFenceInfoExt, + PNext = PresentIdEnabled ? &presentIdInfo : null, + SwapchainCount = 1, + PFences = &presentFence, + }; + + var presentInfo = new PresentInfoKHR + { + SType = StructureType.PresentInfoKhr, + PNext = presentFence.Handle != 0 ? &fenceInfo : PresentIdEnabled ? &presentIdInfo : null, + WaitSemaphoreCount = 1, + PWaitSemaphores = &wait, + SwapchainCount = 1, + PSwapchains = &handle, + PImageIndices = &index, + }; + + // Presenting is a queue operation like any other, so it takes the same + // lock as submission. + Result result; + long waitStart = VulkanStats.WaitStart(); + lock (_context.QueueLock) + { + result = _swapchainApi.QueuePresent(_context.GraphicsQueue, &presentInfo); + } + target.Slot.FinishPresentFence(presentFence, result); + VulkanStats.NoteWait(WaitSite.Present, waitStart); + if (result is Result.Success or Result.SuboptimalKhr) target.Slot.NotePresented(index); + if (result is Result.ErrorOutOfDateKhr or Result.SuboptimalKhr) + { + if (ReferenceEquals(target.Slot, _current)) NeedsRecreation = true; + } + else + { + VulkanResult.Check(result, "vkQueuePresentKHR"); + } + + return presentId; + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + // Teardown, not recreation: everything this chain ever presented must be + // finished before its slots go. + VulkanStats.WaitDeviceIdle(_context.Api, _context.Device); + _retirement.DisposeAll(); + _current?.Dispose(); + _current = null; + // Nothing may be called against these handles again; the backend outlives + // the swapchain (the device disposes it last). + + if (_surface.Handle != 0) + { + _surfaceApi.DestroySurface(_context.Instance, _surface, null); + } + + _swapchainApi.Dispose(); + _surfaceApi.Dispose(); + } +} + +/// +/// The wait stages of the present submission, checked in one place. +/// +/// Submit B waits on two things: the Frame timeline at the value Submit A +/// signalled (the frame image it reads is finished), at COLOR_ATTACHMENT_OUTPUT; +/// and the acquire semaphore at the stage where the swapchain image is first +/// touched: TRANSFER for the flipped blit, COLOR_ATTACHMENT_OUTPUT for a raster +/// path (FSR's final pass). ALL_COMMANDS would also block the barrier and every +/// earlier command on the acquire, which is what the split exists to avoid. +/// +internal static class PresentWaitStages +{ + public const PipelineStageFlags FrameWait = PipelineStageFlags.ColorAttachmentOutputBit; + public const PipelineStageFlags BlitAcquireWait = PipelineStageFlags.TransferBit; + public const PipelineStageFlags RasterAcquireWait = PipelineStageFlags.ColorAttachmentOutputBit; + + /// Throws for a wait stage the present submission must not use. + public static PipelineStageFlags RequireAcquireStage(PipelineStageFlags stage) + { + if (stage != BlitAcquireWait && stage != RasterAcquireWait) + { + throw new ArgumentOutOfRangeException(nameof(stage), stage, + "the present submission waits on the acquire semaphore at TRANSFER or COLOR_ATTACHMENT_OUTPUT only"); + } + return stage; + } +} + +/// +/// The presentation blit: the whole frame, GUI +/// included, renders into the owned default image, and this copies it into the +/// acquired swapchain image, flipped. +/// +/// This inverted blit is the entire Y-flip story for the backend. Everything +/// upstream stays in OpenGL's orientation, which is what keeps intermediate +/// targets and screenshots byte-identical to the GL path; the display wants row 0 +/// at the top, so the source rows are read bottom-to-top exactly once, here. +/// +internal sealed unsafe class BlitPresentPath +{ + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly Func _source; + /// Created at the first record, so a path built for its stage table alone needs no texture table. + private BarrierBatcher? _barriers; + + /// The acquired image's state; reset per frame, since its contents are discarded. + private readonly ResourceStateTracker _swapchainImage = new(1, 1, depth: false); + + public BlitPresentPath(VulkanContext context, TextureManager textures, Func source) + { + _context = context; + _textures = textures; + _source = source; + } + + public PipelineStageFlags AcquireWaitStage => PresentWaitStages.BlitAcquireWait; + + public void Record(CommandBuffer commandBuffer, in PresentTarget target) + { + Image destination = target.Image; + + // Every present leaves the swapchain image in PRESENT_SRC, and nothing + // else writes it, so UNDEFINED discards nothing that matters. The + // destination and the source move in one barrier command. + // The acquire touched it last, at the stage this submission waits on it. + BarrierBatcher barriers = _barriers ??= _textures.CreateBatcher(); + _swapchainImage.Reset((PipelineStageFlags2)(ulong)AcquireWaitStage); + barriers.Require(destination, ImageAspectFlags.ColorBit, _swapchainImage, 0, 1, 0, 1, + ResourceUsage.TransferDst, discard: true); + + VulkanTexture? source = _source(); + if (source != null) _textures.Require(barriers, commandBuffer, source, ResourceUsage.TransferSrc); + barriers.Flush(commandBuffer); + + if (source != null) + { + + var blit = new ImageBlit + { + SrcSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + DstSubresource = new ImageSubresourceLayers(ImageAspectFlags.ColorBit, 0, 0, 1), + }; + // Source Y runs backwards: this is the flip. + blit.SrcOffsets.Element0 = new Offset3D(0, (int)source.Height, 0); + blit.SrcOffsets.Element1 = new Offset3D((int)source.Width, 0, 1); + blit.DstOffsets.Element0 = new Offset3D(0, 0, 0); + blit.DstOffsets.Element1 = new Offset3D((int)target.Extent.Width, (int)target.Extent.Height, 1); + + _context.Api.CmdBlitImage(commandBuffer, + source.Image, ImageLayout.TransferSrcOptimal, + destination, ImageLayout.TransferDstOptimal, + 1, &blit, Filter.Linear); + } + + // TRANSFER_DST (written at TRANSFER) to PRESENT_SRC (BOTTOM_OF_PIPE, no access). + barriers.Require(destination, ImageAspectFlags.ColorBit, _swapchainImage, 0, 1, 0, 1, + ResourceUsage.PresentSrc, discard: false); + barriers.Flush(commandBuffer); + } +} diff --git a/Optimum.Render.Vulkan/Present/SwapchainRetirement.cs b/Optimum.Render.Vulkan/Present/SwapchainRetirement.cs new file mode 100644 index 00000000..35962aec --- /dev/null +++ b/Optimum.Render.Vulkan/Present/SwapchainRetirement.cs @@ -0,0 +1,287 @@ +using System; +using System.Collections.Generic; +using Silk.NET.Vulkan; + +// The Present/ folder follows the plan's layout; the namespace stays Core until +// the renderer is reorganised, like Frame/. +namespace Optimum.Render.Vulkan.Core; + +/// +/// Replaced swapchains wait for presentation completion, not just render completion. +/// Present fences provide a direct completion check when supported. Otherwise a +/// submission waiting on reacquisition of the successor's first presented image +/// provides the completion proof. Resize storms can leave several old chains queued +/// until a successor reaches that point. Render thread only. +/// https://docs.vulkan.org/samples/latest/samples/api/swapchain_recreation/README.html +/// +internal sealed class SwapchainRetirement +{ + private readonly record struct Entry(IDisposable Slot, ulong LastPresentValue, ulong? CompletionValue, Func? PresentsComplete); + + private readonly ITimelineClock _clock; + private readonly List _entries = new(); + + public SwapchainRetirement(ITimelineClock clock) => _clock = clock; + + public int PendingCount => _entries.Count; + + /// Queues a replaced slot (0: never submitted for presentation). + public void Retire(IDisposable slot, ulong lastPresentValue, Func? presentsComplete = null) => + _entries.Add(new Entry(slot, lastPresentValue, lastPresentValue == 0 ? 0UL : null, presentsComplete)); + + /// The successor's first presented image was reacquired, and this + /// submission waits on its acquire semaphore. Completion proves the old + /// chains are no longer being presented. Merely returning from acquire does not. + public void NoteSuccessorReacquired(ulong frameValue) + { + if (frameValue == 0) throw new ArgumentOutOfRangeException(nameof(frameValue)); + for (int i = 0; i < _entries.Count; i++) + { + Entry entry = _entries[i]; + if (!entry.CompletionValue.HasValue && frameValue > entry.LastPresentValue) + _entries[i] = entry with { CompletionValue = frameValue }; + } + } + + /// Destroys, oldest first, every slot whose presentation completion was proven. Returns how many. + public int Collect() + { + if (_entries.Count == 0) return 0; + + ulong completed = _clock.FrameCompleted; + int destroyed = 0; + int kept = 0; + for (int i = 0; i < _entries.Count; i++) + { + Entry entry = _entries[i]; + if (entry.LastPresentValue <= completed && (entry.PresentsComplete != null + ? entry.PresentsComplete() + : entry.CompletionValue is ulong value && value <= completed)) + { + entry.Slot.Dispose(); + destroyed++; + } + else + { + _entries[kept++] = entry; + } + } + _entries.RemoveRange(kept, _entries.Count - kept); + return destroyed; + } + + /// Teardown only, after the GPU finished every submission. + public void DisposeAll() + { + foreach (Entry entry in _entries) entry.Slot.Dispose(); + _entries.Clear(); + } +} + +/// What the present path does with the result of vkAcquireNextImageKHR. +internal enum AcquireAction +{ + /// The image is good; present it. + Present, + /// SUBOPTIMAL: the image is acquired and usable; rebuild before the next acquire. + PresentThenRebuild, + /// OUT_OF_DATE on the first attempt: rebuild now and acquire once more. + RebuildAndRetry, + /// OUT_OF_DATE again after the rebuild: skip presenting this frame, rebuild next frame. + SkipFrame, + /// Anything else (a lost device above all): reported, never a silent skip. + Fail, +} + +/// +/// The swapchain's decisions, as pure functions so they are tested without a +/// window (SwapchainRetirementTests). +/// +internal static class SwapchainPolicy +{ + /// + /// SUBOPTIMAL rebuilds before the next acquire; OUT_OF_DATE rebuilds and + /// re-acquires once; a second OUT_OF_DATE gives up on this frame. + /// + public static AcquireAction OnAcquire(Result result, int attempt) + { + if (result == Result.Success) return AcquireAction.Present; + if (result == Result.SuboptimalKhr) return AcquireAction.PresentThenRebuild; + if (result == Result.ErrorOutOfDateKhr) return attempt == 0 ? AcquireAction.RebuildAndRetry : AcquireAction.SkipFrame; + return AcquireAction.Fail; + } + + /// + /// max(caps.min + 1, mailbox ? 3 : 2), clamped to the surface maximum + /// (0 means unbounded). Mailbox wants a third image so a finished frame can + /// replace the queued one while another is on screen. + /// + public static uint ChooseImageCount(uint capabilitiesMin, uint capabilitiesMax, PresentModeKHR mode) + { + uint wanted = Math.Max(capabilitiesMin + 1, mode == PresentModeKHR.MailboxKhr ? 3u : 2u); + if (capabilitiesMax > 0 && wanted > capabilitiesMax) wanted = capabilitiesMax; + return wanted; + } + + /// A minimised window reports a zero extent; presentation parks until it grows again. + public static bool IsParked(Extent2D extent) => extent.Width == 0 || extent.Height == 0; + + /// + /// With vsync: FIFO, promoted to FIFO_RELAXED once sustained missed vsyncs + /// were seen and the surface offers it (a late frame tears instead of waiting + /// a whole interval). Without vsync: MAILBOX (drops frames, never tears), then + /// IMMEDIATE, then FIFO, the only mode every driver must have. + /// + public static PresentModeKHR ChoosePresentMode(bool vsync, bool relaxedPromoted, IReadOnlyList supported) + { + if (vsync) + { + return relaxedPromoted && Contains(supported, PresentModeKHR.FifoRelaxedKhr) + ? PresentModeKHR.FifoRelaxedKhr + : PresentModeKHR.FifoKhr; + } + if (Contains(supported, PresentModeKHR.MailboxKhr)) return PresentModeKHR.MailboxKhr; + if (Contains(supported, PresentModeKHR.ImmediateKhr)) return PresentModeKHR.ImmediateKhr; + return PresentModeKHR.FifoKhr; + } + + private static bool Contains(IReadOnlyList modes, PresentModeKHR mode) + { + for (int i = 0; i < modes.Count; i++) + { + if (modes[i] == mode) return true; + } + return false; + } +} + +/// +/// A slot's acquire semaphores: imageCount + 1 binary semaphores, taken for +/// each vkAcquireNextImageKHR. +/// +/// A semaphore whose acquire failed is untouched and returns at once +/// (). One whose signal a present submission waits on stays +/// in use until that submission completed on the GPU: a binary semaphore with an +/// uncompleted wait must not be handed to another acquire +/// (VUID-vkAcquireNextImageKHR-semaphore-01779), so it is parked against the +/// submission's Frame value () and reclaimed by a later +/// once the Frame counter passed it. A semaphore whose acquire +/// succeeded but was never submitted is never returned; it dies with the slot. +/// +/// The ring paces on frame n - FramesInFlight, so at most FramesInFlight - 1 present +/// submissions are uncompleted when an acquire starts; imageCount + 1 covers that. +/// +internal sealed class AcquireSemaphoreFreeList +{ + private readonly Stack _free = new(); + private readonly List<(ulong Handle, ulong FrameValue)> _pending = new(); + + public AcquireSemaphoreFreeList(IReadOnlyList handles) + { + for (int i = handles.Count - 1; i >= 0; i--) _free.Push(handles[i]); + Capacity = handles.Count; + } + + public int Capacity { get; } + public int FreeCount => _free.Count; + public int PendingCount => _pending.Count; + + public static int CapacityFor(uint imageCount) => (int)imageCount + 1; + + /// A free semaphore; parked ones whose submission completed (Frame counter ) are reclaimed first when none is free. + public ulong Take(ulong frameCompleted) + { + if (_free.Count == 0) Reclaim(frameCompleted); + if (_free.Count == 0) + { + throw new InvalidOperationException( + "every acquire semaphore of this swapchain is waited on by an uncompleted present submission " + + "or was signalled by an acquire that was never presented"); + } + return _free.Pop(); + } + + /// The acquire failed; the semaphore was never signalled. + public void Return(ulong handle) + { + CheckRoom(); + _free.Push(handle); + } + + /// A submission carrying Frame value waits on the semaphore. + public void ReturnAfter(ulong handle, ulong frameValue) + { + CheckRoom(); + _pending.Add((handle, frameValue)); + } + + private void CheckRoom() + { + if (_free.Count + _pending.Count >= Capacity) throw new InvalidOperationException("acquire semaphore returned twice"); + } + + private void Reclaim(ulong frameCompleted) + { + int kept = 0; + for (int i = 0; i < _pending.Count; i++) + { + if (_pending[i].FrameValue <= frameCompleted) _free.Push(_pending[i].Handle); + else _pending[kept++] = _pending[i]; + } + _pending.RemoveRange(kept, _pending.Count - kept); + } +} + +/// +/// Decides when FIFO should be promoted to FIFO_RELAXED: once enough frames in a +/// window missed their vsync. The refresh interval is estimated as the shortest +/// frame interval seen in the window (under FIFO nothing presents faster than the +/// display), and a frame counts as a miss when it took more than 1.5 of those. +/// A game that is consistently slower than the display never looks like it missed +/// anything, which is intended: relaxed FIFO only helps occasional late frames. +/// +internal sealed class MissedVsyncDetector +{ + public const int Window = 120; + public const int MissesToPromote = 12; + /// Intervals below this are not display refreshes (a burst after a stall). + public const double ShortestPlausibleIntervalMs = 4.0; + + private readonly double[] _intervals = new double[Window]; + private int _count; + private int _next; + + public void Reset() + { + _count = 0; + _next = 0; + } + + /// Adds one present-to-present interval; true once promotion is warranted (then resets). + public bool NoteInterval(double milliseconds) + { + if (!(milliseconds > 0) || double.IsInfinity(milliseconds)) return false; + + _intervals[_next] = milliseconds; + _next = (_next + 1) % Window; + if (_count < Window) _count++; + if (_count < Window) return false; + + double period = double.MaxValue; + for (int i = 0; i < Window; i++) + { + if (_intervals[i] >= ShortestPlausibleIntervalMs && _intervals[i] < period) period = _intervals[i]; + } + if (period == double.MaxValue) return false; + + int misses = 0; + for (int i = 0; i < Window; i++) + { + if (_intervals[i] > period * 1.5) misses++; + } + if (misses < MissesToPromote) return false; + + Reset(); + return true; + } +} diff --git a/Optimum.Render.Vulkan/Shaders/FrameGlobals.cs b/Optimum.Render.Vulkan/Shaders/FrameGlobals.cs new file mode 100644 index 00000000..6e14cb32 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/FrameGlobals.cs @@ -0,0 +1,296 @@ +using System; +using System.Collections.Generic; + +namespace Optimum.Render.Vulkan.Shaders; + +/// +/// The per-frame values every program reads, as one block shared by all of them: +/// descriptor set 0, binding 0. +/// +/// ShaderProgramBase.Use() writes the same values from +/// DefaultShaderUniforms into every program that includes the file that +/// declares them - fog and light, the shadow cascades, the vertex warp, the sky +/// colour, the colour map and the underwater effect - up to 56 writes per program +/// switch. Stored per program, each of those programs carries its own copy and a +/// draw copies all of it into the frame's uniform ring. Stored here, the values +/// live once: a write that changes nothing is a comparison, a write that changes +/// something bumps one version, and every draw in the frame binds the same +/// snapshot until the next change. +/// +/// A name joins the shared block for a program only when that program's +/// Use() really writes it, which is when the program includes the file +/// listed as the name's owner. The GUI program sets lightPosition itself +/// without including fog and light; a program that includes only the fragment +/// half of fog and light declares flatFogDensity but never has it written. +/// Both keep a copy of their own, exactly as on OpenGL, where every program has +/// its own uniform storage. +/// +/// Offsets are fixed for the whole process, independent of which programs exist: +/// arrays are sized for the largest declaration the game can produce (100 dynamic +/// lights is the settings slider's maximum), and a shader that declares a shorter +/// array reads a prefix of the same member. Scalar block layout, as in the +/// per-program block, so the game's packed float arrays land as a memcpy. +/// +internal static class FrameGlobals +{ + public const int Set = 0; + public const int Binding = 0; + public const string BlockTypeName = "OptimumFrameGlobals"; + + /// The dynamic lights slider's maximum; DYNLIGHTS never exceeds it. + public const int MaxDynamicLights = 100; + + private readonly record struct Entry(string Name, string TypeName, int Capacity, string Owner, string? Initializer = null); + + // Owners and initialisers mirror ShaderProgramBase.Use() and the vanilla + // include declarations; shader delivery and draw tests verify the resulting ABI. + private static readonly Entry[] Entries = + { + // fogandlight.fsh + new("zNear", "float", 0, "fogandlight.fsh", "0.3"), + new("zFar", "float", 0, "fogandlight.fsh", "1500.0"), + new("lightPosition", "vec3", 0, "fogandlight.fsh"), + new("shadowIntensity", "float", 0, "fogandlight.fsh", "1"), + new("glitchStrength", "float", 0, "fogandlight.fsh", "0"), + new("psychedelicStrength", "float", 0, "fogandlight.fsh", "0"), + new("shadowMapWidthInv", "float", 0, "fogandlight.fsh"), + new("shadowMapHeightInv", "float", 0, "fogandlight.fsh"), + + // fogandlight.vsh (also the unconditional writer of the view distances) + new("viewDistance", "float", 0, "fogandlight.vsh"), + new("viewDistanceLod0", "float", 0, "fogandlight.vsh"), + new("fogSphereQuantity", "int", 0, "fogandlight.vsh"), + new("pointLightQuantity", "int", 0, "fogandlight.vsh"), + new("flatFogDensity", "float", 0, "fogandlight.vsh"), + new("flatFogStart", "float", 0, "fogandlight.vsh"), + new("glitchStrengthFL", "float", 0, "fogandlight.vsh"), + new("nightVisionStrength", "float", 0, "fogandlight.vsh"), + + // shadowcoords.vsh + new("shadowRangeNear", "float", 0, "shadowcoords.vsh"), + new("shadowRangeFar", "float", 0, "shadowcoords.vsh"), + + // vertexwarp.vsh + new("timeCounter", "float", 0, "vertexwarp.vsh"), + new("windWaveCounter", "float", 0, "vertexwarp.vsh"), + new("windWaveCounterHighFreq", "float", 0, "vertexwarp.vsh"), + new("windSpeed", "float", 0, "vertexwarp.vsh"), + new("waterWaveCounter", "float", 0, "vertexwarp.vsh"), + new("playerpos", "vec3", 0, "vertexwarp.vsh"), + new("globalWarpIntensity", "float", 0, "vertexwarp.vsh"), + new("glitchWaviness", "float", 0, "vertexwarp.vsh", "0"), + new("windWaveIntensity", "float", 0, "vertexwarp.vsh", "1"), + new("waterWaveIntensity", "float", 0, "vertexwarp.vsh", "1"), + new("perceptionEffectId", "int", 0, "vertexwarp.vsh", "1"), + new("perceptionEffectIntensity", "float", 0, "vertexwarp.vsh", "1"), + + // skycolor.fsh (sky and glow are samplers and stay per program) + new("fogWaveCounter", "float", 0, "skycolor.fsh"), + new("sunsetMod", "float", 0, "skycolor.fsh"), + new("ditherSeed", "int", 0, "skycolor.fsh"), + new("horizontalResolution", "int", 0, "skycolor.fsh"), + new("playerToSealevelOffset", "float", 0, "skycolor.fsh"), + + // colormap.vsh + new("seasonRel", "float", 0, "colormap.vsh"), + new("seaLevel", "float", 0, "colormap.vsh"), + new("atlasHeight", "float", 0, "colormap.vsh"), + new("seasonTemperature", "float", 0, "colormap.vsh"), + + // underwatereffects.fsh. frameSize is not here: bilateralblur and the blur + // passes declare a frameSize of their own and write it outside Use(). + new("cameraUnderwater", "float", 0, "underwatereffects.fsh"), + new("waterMurkColor", "vec4", 0, "underwatereffects.fsh"), + + // The large members last, so the scalars above share a few cache lines. + new("toShadowMapSpaceMatrixNear", "mat4", 0, "shadowcoords.vsh"), + new("toShadowMapSpaceMatrixFar", "mat4", 0, "shadowcoords.vsh"), + new("fogSpheres", "float", 3 * 8, "fogandlight.vsh"), + new("colorMapRects", "vec4", 40, "colormap.vsh"), + new("pointLights", "vec3", MaxDynamicLights, "fogandlight.vsh"), + new("pointLightColors", "vec3", MaxDynamicLights, "fogandlight.vsh"), + }; + + private static readonly Dictionary ByName = new(StringComparer.Ordinal); + private static readonly List MemberList = new(); + + /// Size of the shared block in bytes. + public static int BlockSize { get; } + + /// Every member, in layout order. + public static IReadOnlyList Members => MemberList; + + static FrameGlobals() + { + int offset = 0; + foreach (Entry entry in Entries) + { + if (!GlslType.TryParse(entry.TypeName, out GlslType type)) + { + throw new InvalidOperationException("frame global '" + entry.Name + "' has unknown type " + entry.TypeName); + } + + offset = Align(offset, type.Alignment); + var member = new UniformMember + { + Name = entry.Name, + Type = type, + ArrayLength = entry.Capacity, + Offset = offset, + Initializer = entry.Initializer, + }; + member.Size = type.Size * member.ElementCount; + offset += member.Size; + + MemberList.Add(member); + ByName.Add(entry.Name, (member, entry.Owner)); + } + BlockSize = offset; + } + + /// + /// Whether a program's declaration of reads the shared + /// block: the program includes the name's owner, and the declaration agrees + /// with the shared member - the same type, and an array no longer than the + /// member (or not an array where the member is not one). + /// + public static bool TryPlace( + string name, GlslType type, int arrayLength, IReadOnlySet? includes, out UniformMember member) + { + member = null!; + if (includes == null || !ByName.TryGetValue(name, out (UniformMember Member, string Owner) found)) return false; + if (!includes.Contains(found.Owner)) return false; + if (!found.Member.Type.Equals(type)) return false; + + bool fits = found.Member.ArrayLength == 0 + ? arrayLength == 0 + : arrayLength > 0 && arrayLength <= found.Member.ArrayLength; + if (!fits) return false; + + member = found.Member; + return true; + } + + /// The member called , whatever its owner. + public static bool TryGetMember(string name, out UniformMember member) + { + bool known = ByName.TryGetValue(name, out (UniformMember Member, string Owner) found); + member = known ? found.Member : null!; + return known; + } + + /// The include whose block in Use() writes . + public static string? OwnerOf(string name) => ByName.TryGetValue(name, out (UniformMember Member, string Owner) found) ? found.Owner : null; + + /// A shadow of the shared block, pre-filled with the declared defaults. + public static byte[] CreateShadow() + { + var buffer = new byte[BlockSize]; + foreach (UniformMember member in MemberList) + { + if (member.Initializer != null) ProgramInterfaceLayout.WriteInitializer(buffer, member); + } + return buffer; + } + + private static int Align(int value, int alignment) => + alignment <= 1 ? value : (value + alignment - 1) / alignment * alignment; + + // ------------------------------------------------------------ native include + + /// The generated native include (docs/vulkan.md sections 1 and 3). + public const string IncludePath = "sources/shaders-vk/include/frame.glsl"; + + /// The block's instance name in native shaders. + public const string InstanceName = "optimumFrame"; + + /// + /// The macro a native source defines to say the program includes + /// (a game include file name such as fogandlight.vsh) in some stage. + /// + public static string OwnerMacro(string owner) => + "OPTIMUM_FRAME_OWNER_" + owner.Replace('.', '_').ToUpperInvariant(); + + /// Every owner, in the order its first member appears in the block. + public static IReadOnlyList Owners + { + get + { + var owners = new List(); + foreach (Entry entry in Entries) + { + if (!owners.Contains(entry.Owner)) owners.Add(entry.Owner); + } + return owners; + } + } + + /// + /// The text of : the block at its fixed offsets under an instance + /// name, then one group of #define name optimumFrame.name lines per owner. + /// + /// A group is outside the include guard and activates when its owner macro is defined and + /// the group has not been emitted yet, so including the file again after defining another + /// owner macro adds that owner's names. That keeps 's rule in native + /// sources: a name reads the frame block only in a program that includes its owner, and a + /// program that does not keeps a record member of the same name. + /// + public static string GenerateInclude() + { + var text = new System.Text.StringBuilder(); + text.Append(""" + // Generated from Optimum.Render.Vulkan/Shaders/FrameGlobals.cs (FrameGlobals.GenerateInclude). + // Regenerate with FrameGlobals.GenerateInclude; ShaderDeliveryTests verifies the compiled ABI. + // + // The FrameGlobals block (docs/vulkan.md): set 0, binding 0, scalar + // layout, bound with a dynamic offset. Members sit at the offsets the renderer writes. + // + // No member is a global name here. A member is the shared frame value only in a program that + // includes the member's owner file, so every owner has its own group of defines below. An + // owner include (fogandlight.frag.glsl, vertexwarp.glsl, ...) defines its owner macro and + // includes this file, which activates its group. A program whose other stage includes an + // owner defines that owner's macro itself before its includes (for example + // OPTIMUM_FRAME_OWNER_FOGANDLIGHT_VSH in a fragment stage that reads flatFogDensity), and a + // program that includes no owner of a name declares the name in its own record instead. + + #ifndef OPTIMUM_FRAME_GLSL + #define OPTIMUM_FRAME_GLSL + + #extension GL_EXT_scalar_block_layout : require + + #include "bindings.glsl" + + + """); + text.Append("layout(set = OPTIMUM_SET_FRAME, binding = OPTIMUM_BINDING_FRAME_GLOBALS, scalar) uniform ") + .Append(BlockTypeName).Append('\n').Append("{\n"); + foreach (UniformMember member in MemberList) + { + text.Append(" layout(offset = ").Append(member.Offset).Append(") ") + .Append(member.Type.Name).Append(' ').Append(member.Name); + if (member.ArrayLength > 0) text.Append('[').Append(member.ArrayLength).Append(']'); + text.Append(";\n"); + } + text.Append("} ").Append(InstanceName).Append(";\n\n") + .Append("// Block size: ").Append(BlockSize).Append(" bytes.\n\n") + .Append("#endif\n"); + + foreach (string owner in Owners) + { + string names = "OPTIMUM_FRAME_NAMES_" + owner.Replace('.', '_').ToUpperInvariant(); + text.Append('\n') + .Append("// ").Append(owner).Append('\n') + .Append("#if defined(").Append(OwnerMacro(owner)).Append(") && !defined(").Append(names).Append(")\n") + .Append("#define ").Append(names).Append('\n'); + foreach (Entry entry in Entries) + { + if (entry.Owner != owner) continue; + text.Append("#define ").Append(entry.Name).Append(' ') + .Append(InstanceName).Append('.').Append(entry.Name).Append('\n'); + } + text.Append("#endif\n"); + } + + return text.ToString(); + } +} diff --git a/Optimum.Render.Vulkan/Shaders/GlslParser.cs b/Optimum.Render.Vulkan/Shaders/GlslParser.cs new file mode 100644 index 00000000..159e6732 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/GlslParser.cs @@ -0,0 +1,983 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Text; + +namespace Optimum.Render.Vulkan.Shaders; + +internal enum GlslDeclarationKind +{ + /// A loose uniform float x; - GL's default uniform block. + DefaultUniform, + /// A sampler or image handle, which becomes a descriptor. + OpaqueUniform, + /// An explicit layout(std140) uniform Block { ... };. + UniformBlock, + /// An explicit layout(std430) buffer Block { ... };. + StorageBlock, + Input, + Output, + /// Anything the parser deliberately does not touch. + Other +} + +/// One top-level declaration, with the source span it occupies. +internal sealed class GlslDeclaration +{ + public GlslDeclarationKind Kind; + public int Start; + public int Length; + + public string TypeName = ""; + public string Name = ""; + + /// 0 when the declaration is not an array. + public int ArrayLength; + + /// The array size as written, when it did not parse as an integer. + public string? UnresolvedArraySize; + + /// Right-hand side of a default value, or null. GL allows these on + /// uniforms and several shaders rely on them (final.fsh's extraGamma). + public string? Initializer; + + /// Contents of layout(...) as written, or null. + public string? LayoutQualifiers; + + /// + /// Absolute span of the layout(...) clause. When the declaration has + /// none, this is a zero-length span at the point one would be inserted, so + /// the rewriter can treat "replace" and "add" as the same operation. + /// + public int LayoutStart; + public int LayoutLength; + + /// Explicit location from a layout qualifier, else -1. + public int Location = -1; + + /// Interpolation and auxiliary qualifiers preceding the type. + public string Qualifiers = ""; + + /// Absolute start of the storage keyword (uniform, buffer, in, ...), or -1. + public int StorageKeywordStart = -1; + + public int End => Start + Length; +} + +/// The parse of one shader stage. +internal sealed class ParsedShader +{ + public string Source = ""; + public List Declarations = new(); + + /// Span of the #version line, or (-1, 0) when absent. + public int VersionStart = -1; + public int VersionLength; + public int VersionNumber; + + /// Spans of every #extension line. + public List<(int Start, int Length)> ExtensionDirectives = new(); + + /// Span of the identifier main in its function signature. + public int MainNameStart = -1; + + public bool HasMain => MainNameStart >= 0; +} + +/// +/// A deliberately shallow GLSL reader. +/// +/// It understands top-level declarations and nothing else: it tracks brace depth, +/// skips comments and strings, and classifies each statement at depth zero. It +/// never builds an expression tree and never looks inside a function body. +/// +/// That shallowness is the point. The rewriter edits spans of the original source +/// rather than regenerating it, so any construct this parser does not recognise - +/// including whatever a mod author writes - survives verbatim. The failure mode +/// is "left alone", not "mangled". +/// +internal static class GlslParser +{ + public static ParsedShader Parse(string source) + { + var result = new ParsedShader { Source = source }; + int position = 0; + int length = source.Length; + + while (position < length) + { + position = SkipTrivia(source, position); + if (position >= length) break; + + if (source[position] == '#') + { + position = ReadDirective(source, position, result); + continue; + } + + int statementStart = position; + position = ReadTopLevelStatement(source, position, out bool hadBraceBlock, out int mainNameStart); + + if (mainNameStart >= 0 && result.MainNameStart < 0) + { + result.MainNameStart = mainNameStart; + } + + if (position > statementStart) + { + GlslDeclaration? declaration = + Classify(source, statementStart, position - statementStart, hadBraceBlock); + if (declaration != null) + { + result.Declarations.Add(declaration); + } + } + else + { + // Defensive: never spin on an unexpected character. + position = statementStart + 1; + } + } + + return result; + } + + // ------------------------------------------------------------------ scanning + + private static int SkipTrivia(string source, int position) + { + int length = source.Length; + while (position < length) + { + char c = source[position]; + if (c == '/' && position + 1 < length) + { + if (source[position + 1] == '/') + { + while (position < length && source[position] != '\n') position++; + continue; + } + if (source[position + 1] == '*') + { + position += 2; + while (position + 1 < length && !(source[position] == '*' && source[position + 1] == '/')) + { + position++; + } + position = Math.Min(position + 2, length); + continue; + } + } + if (!char.IsWhiteSpace(c)) break; + position++; + } + return position; + } + + /// Reads a preprocessor line, recording #version and #extension. + private static int ReadDirective(string source, int position, ParsedShader result) + { + int start = position; + int length = source.Length; + + // A directive can be continued with a trailing backslash. + while (position < length) + { + if (source[position] == '\\' && position + 1 < length && + (source[position + 1] == '\n' || source[position + 1] == '\r')) + { + position += 2; + continue; + } + if (source[position] == '\n') break; + position++; + } + + string line = source.Substring(start, position - start); + string trimmed = line.TrimStart('#', ' ', '\t'); + + if (trimmed.StartsWith("version", StringComparison.Ordinal)) + { + result.VersionStart = start; + result.VersionLength = position - start; + foreach (string token in trimmed.Split(new[] { ' ', '\t' }, StringSplitOptions.RemoveEmptyEntries)) + { + if (int.TryParse(token, NumberStyles.Integer, CultureInfo.InvariantCulture, out int version)) + { + result.VersionNumber = version; + break; + } + } + } + else if (trimmed.StartsWith("extension", StringComparison.Ordinal)) + { + result.ExtensionDirectives.Add((start, position - start)); + } + + return position; + } + + /// + /// Reads one top-level statement. A declaration ends at its semicolon; a + /// function definition ends at the closing brace of its body. A block + /// declaration has both - uniform B { ... } name; - and the trailing + /// semicolon is included. + /// + private static int ReadTopLevelStatement(string source, int position, out bool hadBraceBlock, out int mainNameStart) + { + int length = source.Length; + int depth = 0; + hadBraceBlock = false; + mainNameStart = -1; + + int lastIdentifierStart = -1; + int lastIdentifierLength = 0; + + while (position < length) + { + position = SkipTrivia(source, position); + if (position >= length) break; + + char c = source[position]; + + // Read a whole identifier in one go. Accumulating character by + // character across the trivia skip would run "void main" together + // into a single nine-character token and lose the function name. + if (IsIdentifierStart(c)) + { + int identifierStart = position; + while (position < length && IsIdentifierPart(source[position])) position++; + lastIdentifierStart = identifierStart; + lastIdentifierLength = position - identifierStart; + continue; + } + + if (c == '{') + { + // The identifier immediately before a top-level '(' ... '{' is the + // function name. Only "main" matters, and only at depth 0. + if (depth == 0) + { + hadBraceBlock = true; + } + depth++; + position++; + continue; + } + + if (c == '}') + { + depth--; + position++; + if (depth <= 0) + { + int after = SkipTrivia(source, position); + if (after < length && source[after] == ';') + { + return after + 1; + } + return position; + } + continue; + } + + if (c == '(' && depth == 0 && lastIdentifierLength == 4 && + string.CompareOrdinal(source, lastIdentifierStart, "main", 0, 4) == 0) + { + mainNameStart = lastIdentifierStart; + } + + if (c == ';' && depth == 0) + { + return position + 1; + } + + position++; + } + + return position; + } + + private static bool IsIdentifierStart(char c) => char.IsLetter(c) || c == '_'; + + private static bool IsIdentifierPart(char c) => char.IsLetterOrDigit(c) || c == '_'; + + // ---------------------------------------------------------------- classifying + + /// + /// Qualifiers that may sit between a layout clause and the storage keyword. + /// The parser steps over them to reach the part it cares about. + /// + /// The memory qualifiers matter as much as the interpolation ones: the chunk + /// shaders declare readonly buffer faceDataBuf, and failing to step + /// over readonly leaves the storage block unrecognised, which lands it + /// in the wrong descriptor set. + /// + private static readonly string[] SkippableQualifiers = + { + "flat", "smooth", "noperspective", "centroid", "sample", "invariant", "precise", + "highp", "mediump", "lowp", + "readonly", "writeonly", "coherent", "volatile", "restrict", + }; + + private static GlslDeclaration? Classify(string source, int start, int length, bool hadBraceBlock) + { + string text = source.Substring(start, length); + string stripped = StripComments(text); + + var declaration = new GlslDeclaration { Start = start, Length = length }; + + int cursor = 0; + // Default the insertion point to the head of the declaration, so a + // declaration with no layout clause still has a valid place to gain one. + declaration.LayoutStart = start; + declaration.LayoutLength = 0; + + ReadLayoutInto(stripped, ref cursor, declaration, start); + + var qualifiers = new List(); + string? storage = null; + + while (true) + { + int save = cursor; + string? word = ReadIdentifier(stripped, ref cursor); + if (word == null) { cursor = save; break; } + + if (word == "uniform" || word == "buffer" || word == "in" || word == "out" || + word == "attribute" || word == "varying" || word == "shared") + { + storage = word; + declaration.StorageKeywordStart = start + cursor - word.Length; + break; + } + + if (Array.IndexOf(SkippableQualifiers, word) >= 0) + { + qualifiers.Add(word); + continue; + } + + if (word == "layout") + { + cursor = save; + ReadLayoutInto(stripped, ref cursor, declaration, start); + continue; + } + + // A const, a struct, a function, a plain global: not ours. + cursor = save; + break; + } + + declaration.Qualifiers = string.Join(" ", qualifiers); + + if (storage == null) + { + declaration.Kind = GlslDeclarationKind.Other; + return declaration; + } + + // An interface block: "uniform Name { ... }" / "buffer Name { ... }". + if (hadBraceBlock && (storage == "uniform" || storage == "buffer")) + { + declaration.Kind = storage == "uniform" + ? GlslDeclarationKind.UniformBlock + : GlslDeclarationKind.StorageBlock; + declaration.Name = ReadIdentifier(stripped, ref cursor) ?? ""; + return declaration; + } + + if (hadBraceBlock) + { + declaration.Kind = GlslDeclarationKind.Other; + return declaration; + } + + string? typeName = ReadIdentifier(stripped, ref cursor); + if (typeName == null) + { + declaration.Kind = GlslDeclarationKind.Other; + return declaration; + } + declaration.TypeName = typeName; + + // C-style array-on-the-type: "uniform vec3[64] samples;" (ssao.fsh). + ReadArraySuffix(stripped, ref cursor, declaration); + + string? name = ReadIdentifier(stripped, ref cursor); + if (name == null) + { + declaration.Kind = GlslDeclarationKind.Other; + return declaration; + } + declaration.Name = name; + + // Array-on-the-name: "uniform vec3 pointLights[100];". + ReadArraySuffix(stripped, ref cursor, declaration); + + SkipSpace(stripped, ref cursor); + if (cursor < stripped.Length && stripped[cursor] == '=') + { + cursor++; + int initializerStart = cursor; + int end = stripped.IndexOf(';', cursor); + if (end < 0) end = stripped.Length; + declaration.Initializer = stripped.Substring(initializerStart, end - initializerStart).Trim(); + cursor = end; + } + + // A comma-separated declaration list ("uniform float a, b;") is legal GLSL + // but appears nowhere in this game or its shaders. Leaving it alone is + // safer than half-handling it: it will fail to compile with a clear + // message rather than silently losing a uniform. + SkipSpace(stripped, ref cursor); + if (cursor < stripped.Length && stripped[cursor] == ',') + { + declaration.Kind = GlslDeclarationKind.Other; + return declaration; + } + + declaration.Kind = storage switch + { + "uniform" => GlslType.IsOpaqueTypeName(typeName) + ? GlslDeclarationKind.OpaqueUniform + : GlslDeclarationKind.DefaultUniform, + "in" or "attribute" => GlslDeclarationKind.Input, + "out" or "varying" => GlslDeclarationKind.Output, + _ => GlslDeclarationKind.Other, + }; + + return declaration; + } + + private static void ReadArraySuffix(string text, ref int cursor, GlslDeclaration declaration) + { + SkipSpace(text, ref cursor); + if (cursor >= text.Length || text[cursor] != '[') return; + + int close = text.IndexOf(']', cursor); + if (close < 0) return; + + string inside = text.Substring(cursor + 1, close - cursor - 1).Trim(); + cursor = close + 1; + + if (TryEvaluateConstantInt(inside, out int size)) + { + declaration.ArrayLength = size; + } + else + { + declaration.UnresolvedArraySize = inside; + } + } + + /// + /// Evaluates an integer constant expression from an array size. + /// + /// GLSL permits any constant expression there and the shaders use it - + /// fogandlight.vsh declares uniform vec4 fogSpheres[3 * 8];. By this + /// point the preprocessor has already substituted every macro, so what is + /// left is arithmetic over literals. + /// + internal static bool TryEvaluateConstantInt(string expression, out int value) + { + int cursor = 0; + value = 0; + + if (!TryParseAdditive(expression, ref cursor, out int result)) return false; + + SkipSpace(expression, ref cursor); + if (cursor != expression.Length) return false; + + value = result; + return true; + } + + private static bool TryParseAdditive(string text, ref int cursor, out int value) + { + value = 0; + if (!TryParseMultiplicative(text, ref cursor, out int left)) return false; + + while (true) + { + SkipSpace(text, ref cursor); + if (cursor >= text.Length) break; + + char op = text[cursor]; + if (op != '+' && op != '-') break; + + cursor++; + if (!TryParseMultiplicative(text, ref cursor, out int right)) return false; + left = op == '+' ? left + right : left - right; + } + + value = left; + return true; + } + + private static bool TryParseMultiplicative(string text, ref int cursor, out int value) + { + value = 0; + if (!TryParseUnary(text, ref cursor, out int left)) return false; + + while (true) + { + SkipSpace(text, ref cursor); + if (cursor >= text.Length) break; + + char op = text[cursor]; + if (op != '*' && op != '/' && op != '%') break; + + cursor++; + if (!TryParseUnary(text, ref cursor, out int right)) return false; + if (op != '*' && right == 0) return false; + + left = op switch + { + '*' => left * right, + '/' => left / right, + _ => left % right, + }; + } + + value = left; + return true; + } + + private static bool TryParseUnary(string text, ref int cursor, out int value) + { + value = 0; + SkipSpace(text, ref cursor); + if (cursor >= text.Length) return false; + + char c = text[cursor]; + if (c == '+' || c == '-') + { + cursor++; + if (!TryParseUnary(text, ref cursor, out int inner)) return false; + value = c == '-' ? -inner : inner; + return true; + } + + if (c == '(') + { + cursor++; + if (!TryParseAdditive(text, ref cursor, out int inner)) return false; + SkipSpace(text, ref cursor); + if (cursor >= text.Length || text[cursor] != ')') return false; + cursor++; + value = inner; + return true; + } + + if (!char.IsDigit(c)) return false; + + int start = cursor; + while (cursor < text.Length && char.IsDigit(text[cursor])) cursor++; + + // A trailing 'u' suffix is legal on an integer literal. + if (cursor < text.Length && (text[cursor] == 'u' || text[cursor] == 'U')) cursor++; + + return int.TryParse( + text.AsSpan(start, cursor - start).TrimEnd('u').TrimEnd('U'), + NumberStyles.Integer, CultureInfo.InvariantCulture, out value); + } + + /// + /// Reads a layout(...) clause if one is present and records both its + /// contents and its absolute span on the declaration. + /// + private static void ReadLayoutInto(string text, ref int cursor, GlslDeclaration declaration, int declarationStart) + { + SkipSpace(text, ref cursor); + int clauseStart = cursor; + + string? qualifiers = ReadLayoutQualifier(text, ref cursor); + if (qualifiers == null) return; + + declaration.LayoutQualifiers = qualifiers; + declaration.LayoutStart = declarationStart + clauseStart; + declaration.LayoutLength = cursor - clauseStart; + declaration.Location = ReadLocation(qualifiers); + } + + private static string? ReadLayoutQualifier(string text, ref int cursor) + { + int save = cursor; + SkipSpace(text, ref cursor); + if (!MatchWord(text, ref cursor, "layout")) { cursor = save; return null; } + + SkipSpace(text, ref cursor); + if (cursor >= text.Length || text[cursor] != '(') { cursor = save; return null; } + + int depth = 0; + int start = cursor + 1; + while (cursor < text.Length) + { + if (text[cursor] == '(') depth++; + else if (text[cursor] == ')') + { + depth--; + if (depth == 0) + { + string inside = text.Substring(start, cursor - start); + cursor++; + return inside; + } + } + cursor++; + } + + cursor = save; + return null; + } + + private static int ReadLocation(string qualifiers) + { + foreach (string part in qualifiers.Split(',')) + { + int equals = part.IndexOf('='); + if (equals < 0) continue; + if (part.AsSpan(0, equals).Trim().SequenceEqual("location") && + int.TryParse(part.AsSpan(equals + 1).Trim(), NumberStyles.Integer, + CultureInfo.InvariantCulture, out int location)) + { + return location; + } + } + return -1; + } + + private static bool MatchWord(string text, ref int cursor, string word) + { + if (cursor + word.Length > text.Length) return false; + if (string.CompareOrdinal(text, cursor, word, 0, word.Length) != 0) return false; + int after = cursor + word.Length; + if (after < text.Length && (char.IsLetterOrDigit(text[after]) || text[after] == '_')) return false; + cursor = after; + return true; + } + + private static string? ReadIdentifier(string text, ref int cursor) + { + SkipSpace(text, ref cursor); + if (cursor >= text.Length || !IsIdentifierStart(text[cursor])) return null; + + int start = cursor; + while (cursor < text.Length && (char.IsLetterOrDigit(text[cursor]) || text[cursor] == '_')) cursor++; + return text.Substring(start, cursor - start); + } + + private static void SkipSpace(string text, ref int cursor) + { + while (cursor < text.Length && char.IsWhiteSpace(text[cursor])) cursor++; + } + + /// + /// Blanks comments while preserving offsets, so spans stay valid. The + /// production path sees preprocessed source with comments already gone; this + /// keeps the parser usable on raw source in tests and on mod shaders that + /// reach it by another route. + /// + internal static string StripComments(string text) + { + var builder = new StringBuilder(text); + int i = 0; + while (i < text.Length) + { + if (text[i] == '/' && i + 1 < text.Length) + { + if (text[i + 1] == '/') + { + while (i < text.Length && text[i] != '\n') builder[i++] = ' '; + continue; + } + if (text[i + 1] == '*') + { + int end = text.IndexOf("*/", i + 2, StringComparison.Ordinal); + end = end < 0 ? text.Length : end + 2; + while (i < end) + { + if (text[i] != '\n') builder[i] = ' '; + i++; + } + continue; + } + } + i++; + } + return builder.ToString(); + } +} + +/// +/// Renames identifiers that GLSL 330 allows but GLSL 450 reserves. +/// +/// Vulkan requires #version 450, and the language gained keywords between +/// the two versions. A shader that used one of them as an ordinary variable name +/// compiled fine against 330 and becomes a syntax error at 450 - +/// ssao.fsh has a local called sample, which 4.00 turned into an +/// interpolation qualifier. +/// +/// The rename is a token-level substitution, so it catches declarations and uses +/// alike without needing to understand the code around them. +/// +internal static class GlslReservedWords +{ + private const string Prefix = "_optimum_kw_"; + + /// + /// Words reserved by 4.x that a 330 shader could legitimately have used as a + /// name. + /// + /// buffer and shared are deliberately absent. Both became + /// storage qualifiers in 4.30 and both are used as qualifiers by these + /// shaders - chunkopaque.vsh declares readonly buffer faceDataBuf - + /// so renaming them would break the declaration this backend depends on. A + /// 330 shader using either as a variable name is possible in principle and + /// would fail to compile with a clear message; that is the better trade. + /// + private static readonly string[] Reserved = + { + "sample", "patch", "subroutine", "precise", + "resource", "filter", "active", "common", "partition", "superp", + "input", "output", + }; + + /// + /// Built-ins GL and Vulkan spell differently. + /// + /// The values also differ in principle - gl_VertexIndex counts from + /// the draw's vertex offset and gl_InstanceIndex from its first + /// instance, where the GL originals count from zero - but this client issues + /// no draw with a non-zero base vertex or first instance, so the two agree + /// everywhere they are used. + /// + private static readonly (string From, string To)[] BuiltinRenames = + { + ("gl_VertexID", "gl_VertexIndex"), + ("gl_InstanceID", "gl_InstanceIndex"), + }; + + private static readonly Dictionary Renames = BuildRenames(); + + private static Dictionary BuildRenames() + { + var renames = new Dictionary(StringComparer.Ordinal); + foreach (string word in Reserved) renames[word] = Prefix + word; + foreach ((string from, string to) in BuiltinRenames) renames[from] = to; + return renames; + } + + /// Longest key, used to skip identifiers that cannot match. + private static readonly int LongestRename = MaxKeyLength(); + + private static int MaxKeyLength() + { + int longest = 0; + foreach (string key in Renames.Keys) longest = Math.Max(longest, key.Length); + return longest; + } + + /// + /// Returns the source with reserved identifiers renamed, or the original + /// string when nothing needed changing. + /// + public static string Rename(string source) + { + if (string.IsNullOrEmpty(source)) return source; + + StringBuilder? builder = null; + int copiedTo = 0; + int position = 0; + int length = source.Length; + + while (position < length) + { + char c = source[position]; + + // Line comments and block comments are gone after preprocessing, but + // this class is cheap to make safe against raw source too. + if (c == '/' && position + 1 < length) + { + if (source[position + 1] == '/') + { + while (position < length && source[position] != '\n') position++; + continue; + } + if (source[position + 1] == '*') + { + int end = source.IndexOf("*/", position + 2, StringComparison.Ordinal); + position = end < 0 ? length : end + 2; + continue; + } + } + + if (!IsIdentifierStart(c)) + { + position++; + continue; + } + + int start = position; + while (position < length && IsIdentifierPart(source[position])) position++; + + int wordLength = position - start; + if (wordLength > LongestRename) continue; + + string word = source.Substring(start, wordLength); + if (!Renames.TryGetValue(word, out string? replacement)) continue; + + builder ??= new StringBuilder(length + 64); + builder.Append(source, copiedTo, start - copiedTo); + builder.Append(replacement); + copiedTo = position; + } + + if (builder == null) return source; + + builder.Append(source, copiedTo, length - copiedTo); + return builder.ToString(); + } + + private static bool IsIdentifierStart(char c) => char.IsLetter(c) || c == '_'; + private static bool IsIdentifierPart(char c) => char.IsLetterOrDigit(c) || c == '_'; +} + +/// +/// The GLSL types that can appear in a uniform declaration, with their scalar +/// block layout sizes. +/// +/// Scalar layout (GL_EXT_scalar_block_layout) aligns every aggregate to its +/// component's natural alignment rather than rounding up to 16 bytes. For the +/// 32-bit types this game uses that means alignment is always 4 and size is +/// simply the component count times 4 - which is exactly how a tightly packed +/// float[] is laid out on the CPU. +/// +/// That equivalence is the whole reason this backend asks for scalar layout. +/// The game feeds uniforms as raw arrays: Uniforms3("pointLights", count, +/// float[]) sends count * 3 floats with no padding, and +/// UniformMatrices4x3 sends 12 floats per matrix. Under std140 a +/// vec3[] has a 16-byte stride and a mat4x3[] a 64-byte one, so +/// every array upload would need re-striding on the CPU. Under scalar layout the +/// setter is a memcpy at a recorded offset. +/// +internal readonly struct GlslType : IEquatable +{ + /// The name as written in the shader, e.g. "vec3", "mat4x3". + public string Name { get; } + + /// Columns for a matrix; 1 for scalars and vectors. + public int Columns { get; } + + /// Components per column: 3 for vec3 and for a column of mat4x3. + public int Rows { get; } + + /// Bytes per scalar component. 4 for every type this game uses. + public int ScalarSize { get; } + + /// + /// True for samplers and images: opaque handles that live in descriptors, not + /// in the uniform block. + /// + public bool IsOpaque { get; } + + private GlslType(string name, int columns, int rows, int scalarSize, bool isOpaque) + { + Name = name; + Columns = columns; + Rows = rows; + ScalarSize = scalarSize; + IsOpaque = isOpaque; + } + + /// Total components, e.g. 12 for mat4x3. + public int ComponentCount => Columns * Rows; + + /// Size of one element in bytes under scalar layout. + public int Size => ComponentCount * ScalarSize; + + /// + /// Alignment under scalar layout: the component's own alignment, never + /// rounded up. This is what keeps array strides tight. + /// + public int Alignment => ScalarSize; + + public bool IsMatrix => Columns > 1; + + public bool Equals(GlslType other) => Name == other.Name; + public override bool Equals(object? obj) => obj is GlslType other && Equals(other); + public override int GetHashCode() => Name?.GetHashCode(StringComparison.Ordinal) ?? 0; + public override string ToString() => Name; + + private static GlslType Numeric(string name, int columns, int rows) => + new(name, columns, rows, 4, isOpaque: false); + + private static GlslType Opaque(string name) => + new(name, 1, 1, 0, isOpaque: true); + + private static readonly Dictionary ByName = BuildTable(); + + private static Dictionary BuildTable() + { + var table = new Dictionary(StringComparer.Ordinal); + + void Add(GlslType type) => table[type.Name] = type; + + // Scalars. bool is 4 bytes in a uniform block, as in GL. + Add(Numeric("float", 1, 1)); + Add(Numeric("int", 1, 1)); + Add(Numeric("uint", 1, 1)); + Add(Numeric("bool", 1, 1)); + + // Vectors. + for (int n = 2; n <= 4; n++) + { + Add(Numeric("vec" + n, 1, n)); + Add(Numeric("ivec" + n, 1, n)); + Add(Numeric("uvec" + n, 1, n)); + Add(Numeric("bvec" + n, 1, n)); + } + + // Matrices. GLSL matCxR is C columns of R rows, and matN is matNxN. + for (int columns = 2; columns <= 4; columns++) + { + Add(Numeric("mat" + columns, columns, columns)); + for (int rows = 2; rows <= 4; rows++) + { + Add(Numeric($"mat{columns}x{rows}", columns, rows)); + } + } + + // Opaque handles. Only the ones the game and its shaders actually use, + // plus the obvious neighbours so a mod shader is not rejected for using + // a sampler type this list happened to omit. + foreach (string sampler in new[] + { + "sampler1D", "sampler2D", "sampler3D", "samplerCube", + "sampler2DShadow", "sampler1DShadow", "samplerCubeShadow", + "sampler2DArray", "sampler2DArrayShadow", "sampler1DArray", + "sampler2DMS", "sampler2DMSArray", "samplerBuffer", + "isampler2D", "isampler3D", "isamplerCube", "isampler2DArray", + "usampler2D", "usampler3D", "usamplerCube", "usampler2DArray", + "image2D", "image3D", "imageCube", "image2DArray", + }) + { + Add(Opaque(sampler)); + } + + return table; + } + + /// + /// Resolves a type name. Returns false for anything unrecognised - a struct, + /// or a type this table does not model - which the rewriter treats as a + /// declaration to leave alone rather than an error. + /// + public static bool TryParse(string name, out GlslType type) => ByName.TryGetValue(name, out type); + + /// True for a name that is a sampler or image handle. + public static bool IsOpaqueTypeName(string name) => + ByName.TryGetValue(name, out GlslType type) && type.IsOpaque; +} diff --git a/Optimum.Render.Vulkan/Shaders/NativeShaderBuilder.cs b/Optimum.Render.Vulkan/Shaders/NativeShaderBuilder.cs new file mode 100644 index 00000000..7c2e1cff --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/NativeShaderBuilder.cs @@ -0,0 +1,761 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.IO; +using System.Linq; +using System.Security.Cryptography; +using System.Text; +using System.Text.RegularExpressions; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// What one build produced: the manifest, the SPIR-V files by name, and every error. +internal sealed class NativeShaderBuildResult +{ + public NativeShaderManifest Manifest = new(); + /// SPIR-V file name (relative to the manifest directory) to bytes. + public SortedDictionary Files = new(StringComparer.Ordinal); + public List Errors = new(); + public bool Success => Errors.Count == 0; +} + +/// +/// The offline native shader compiler behind tools/shader-compiler +/// (docs/vulkan.md sections 5 and 6). +/// +/// For every <program>.glsl in the source directory it resolves +/// #includes (the including file's directory first, then include/), finds the variant +/// axes the source branches on, compiles every combination twice - the shipped optimised module and +/// an unoptimised twin that keeps names and declarations - reflects both, checks the result against +/// the set convention, and records it in a . +/// +internal sealed class NativeShaderBuilder +{ + /// The define symbols that stay compile-time variants (contract section 5), sorted. + public static readonly string[] VariantAxes = + { + "ALLOWDEPTHOFFSET", "GBUFFER", "GLOWSUB", "GREEDYMESH", "TAAMOTION", "USEOIT", "USESSBO", "VEC3SCALE", + }; + + public const string IncludeDirectoryName = "include"; + + /// + /// Declares a sampler slot in the push block: OPTIMUM_SAMPLER_SLOT(sampler2D, terrainTex) + /// expands to uint terrainTex (bindings.glsl). A SPIR-V uint does not say which + /// sampler type it indexes, so the builder reads the declaration from the source and checks it + /// against the array the shipped module actually indexes. + /// + public const string SamplerSlotMacro = "OPTIMUM_SAMPLER_SLOT"; + + private const int MaxIncludeDepth = 16; + + private static readonly Regex IncludeDirective = new(@"^[ \t]*#[ \t]*include[ \t]+[""<]([^"">]+)["">][^\r\n]*", RegexOptions.Multiline); + private static readonly Regex ConditionalDirective = new(@"^[ \t]*#[ \t]*(if|elif|ifdef|ifndef)\b([^\r\n]*)", RegexOptions.Multiline); + private static readonly Regex Identifier = new(@"\b[A-Za-z_][A-Za-z0-9_]*\b"); + private static readonly Regex DefinedAxis = new(@"\bdefined\s*\(?\s*([A-Za-z_][A-Za-z0-9_]*)"); + private static readonly Regex SamplerSlot = new(@"^(?![ \t]*#)[^\r\n]*?\bOPTIMUM_SAMPLER_SLOT\s*\(\s*(\w+)\s*,\s*(\w+)\s*\)", RegexOptions.Multiline); + private static readonly Regex SamplerSlotAnywhere = new(@"\bOPTIMUM_SAMPLER_SLOT\s*\(\s*(\w+)\s*,\s*(\w+)\s*\)"); + + private readonly ShaderCompiler _compiler; + + public NativeShaderBuilder(ShaderCompiler compiler) + { + _compiler = compiler; + } + + // ------------------------------------------------------------------ build + + /// Builds every program in , or only . + public NativeShaderBuildResult Build(string sourceDirectory, string? onlyProgram = null) + { + var result = new NativeShaderBuildResult(); + result.Manifest.Toolchain = _compiler.Identity; + + if (!Directory.Exists(sourceDirectory)) + { + result.Errors.Add("source directory not found: " + sourceDirectory); + return result; + } + + List programs = DiscoverPrograms(sourceDirectory, result.Errors); + if (onlyProgram != null) + { + if (!programs.Contains(onlyProgram)) + { + result.Errors.Add("no program '" + onlyProgram + "' (needs " + onlyProgram + ".glsl)"); + return result; + } + programs = new List { onlyProgram }; + } + + foreach (string program in programs) + { + NativeProgram? built = BuildProgram(sourceDirectory, program, result); + if (built != null) result.Manifest.Programs.Add(built); + } + return result; + } + + /// Each root GLSL file owns both stages and their shared interface. + public static List DiscoverPrograms(string sourceDirectory, List errors) + { + var programs = Directory.GetFiles(sourceDirectory, "*.glsl") + .Select(Path.GetFileNameWithoutExtension).ToList(); + programs.Sort(StringComparer.Ordinal); + return programs; + } + + private NativeProgram? BuildProgram(string sourceDirectory, string name, NativeShaderBuildResult result) + { + var stages = new (string Extension, string StageName, EnumShaderType Type)[] + { + ("vert", "vertex", EnumShaderType.VertexShader), + ("frag", "fragment", EnumShaderType.FragmentShader), + }; + + var includes = new SortedSet(StringComparer.Ordinal); + string source; + string fileName = name + ".glsl"; + try + { + source = ExpandIncludes(Path.Combine(sourceDirectory, fileName), sourceDirectory, includes); + } + catch (InvalidDataException error) + { + result.Errors.Add(name + ": " + error.Message); + return null; + } + + int errorsBefore = result.Errors.Count; + List axes = AxesOf(name, new[] { source }, result.Errors); + Dictionary slotTypes = SamplerSlotDeclarations(name, new[] { source }, result.Errors); + if (result.Errors.Count > errorsBefore) return null; + + var program = new NativeProgram { Name = name, Axes = axes }; + for (int combination = 0; combination < 1 << axes.Count; combination++) + { + var values = new Dictionary(StringComparer.Ordinal); + var prefix = new StringBuilder(); + for (int a = 0; a < axes.Count; a++) + { + int value = (combination >> (axes.Count - 1 - a)) & 1; + values[axes[a]] = value; + prefix.Append("#define ").Append(axes[a]).Append(' ').Append(value.ToString(CultureInfo.InvariantCulture)).Append('\n'); + } + string key = NativeShaderManifest.VariantKey(axes, values); + string label = name + (key.Length > 0 ? " [" + key + "]" : ""); + string fileStem = name + string.Concat(axes.Select(axis => "." + axis + values[axis].ToString(CultureInfo.InvariantCulture))); + + var shipped = new SpirvModuleReflection[stages.Length]; + var declared = new SpirvModuleReflection[stages.Length]; + var variant = new NativeVariant { Key = key }; + bool compiled = true; + for (int i = 0; i < stages.Length; i++) + { + string stageDefine = i == 0 ? "OPTIMUM_VERTEX" : "OPTIMUM_FRAGMENT"; + string code = ShaderCompiler.SplicePrefix(source, "#define " + stageDefine + " 1\n" + prefix); + ShaderCompileResult optimised = _compiler.Compile(code, fileName, stages[i].Type); + ShaderCompileResult reflection = optimised.Success + ? _compiler.CompileForReflection(code, fileName, stages[i].Type) + : optimised; + if (!optimised.Success || !reflection.Success) + { + result.Errors.Add(label + " " + fileName + ": " + (optimised.Error ?? reflection.Error)); + compiled = false; + continue; + } + + shipped[i] = SpirvReflection.Reflect(optimised.Spirv); + declared[i] = SpirvReflection.Reflect(reflection.Spirv); + + string spirvName = fileStem + "." + stages[i].Extension + ".spv"; + result.Files[spirvName] = optimised.Spirv; + variant.Stages.Add(new NativeStage + { + Stage = stages[i].StageName, + Source = fileName, + Spirv = spirvName, + Sha256 = Convert.ToHexStringLower(SHA256.HashData(optimised.Spirv)), + }); + } + if (!compiled) continue; + + if (DescribeVariant(label, variant, declared[0], declared[1], shipped[0], shipped[1], slotTypes, includes, result.Errors)) + { + program.Variants.Add(variant); + } + } + + program.Variants.Sort((a, b) => string.CompareOrdinal(a.Key, b.Key)); + return program; + } + + /// Fills the variant's reflected fields and checks them; false when anything is wrong. + private static bool DescribeVariant( + string label, NativeVariant variant, + SpirvModuleReflection vertex, SpirvModuleReflection fragment, + SpirvModuleReflection shippedVertex, SpirvModuleReflection shippedFragment, + Dictionary slotTypes, IReadOnlySet includes, List errors) + { + int errorsBefore = errors.Count; + + // Both stages share declarations in the program source, so their layouts must agree. + SpirvBlock? push = Agree(label, "push block", vertex.PushConstants, fragment.PushConstants, errors); + if (push != null && push.Size > SetConvention.PushConstantBytes) + { + errors.Add(label + ": push block is " + push.Size + " B, the limit is " + SetConvention.PushConstantBytes); + } + variant.Push = ToNative(push); + + foreach (SpirvModuleReflection module in new[] { vertex, fragment }) + { + foreach (SpirvDescriptorBinding binding in module.Bindings) CheckConvention(label, binding, errors); + } + + SpirvBlock? record = Agree(label, "program record", + RecordOf(vertex), RecordOf(fragment), errors); + variant.Record = ToNative(record); + + // Sampler slots: uint push members declared with OPTIMUM_SAMPLER_SLOT, first in the block. + if (push != null) + { + bool seenOther = false; + for (int index = 0; index < push.Members.Count; index++) + { + SpirvBlockMember member = push.Members[index]; + var indexed = new SortedSet<(int Set, int Binding)>(); + foreach (SpirvModuleReflection module in new[] { shippedVertex, shippedFragment }) + { + if (module.PushMemberIndexes.TryGetValue(index, out SortedSet<(int Set, int Binding)>? found)) indexed.UnionWith(found); + } + + if (!slotTypes.TryGetValue(member.Name, out string? glslType)) + { + seenOther = true; + if (indexed.Any(pair => pair.Set == SetConvention.TextureSet)) + { + errors.Add(label + ": push member '" + member.Name + "' indexes a texture array but is not declared with " + SamplerSlotMacro); + } + continue; + } + + if (member.GlslType != "uint" || member.ArrayLength != 0) + { + errors.Add(label + ": sampler slot '" + member.Name + "' must be a uint, is " + member.GlslType); + continue; + } + if (seenOther) + { + errors.Add(label + ": sampler slot '" + member.Name + "' follows a non-slot push member; slots come first"); + } + + SetConvention.Binding array = Array.Find(SetConvention.TextureArrays, b => b.GlslType == glslType); + foreach ((int set, int binding) in indexed) + { + if (set != SetConvention.TextureSet || binding != array.Value) + { + errors.Add(label + ": sampler slot '" + member.Name + "' is declared " + glslType + " (" + array.Name + + ", set 1 binding " + array.Value + ") but indexes set " + set + " binding " + binding); + } + } + + variant.Samplers.Add(new NativeSampler + { + Name = member.Name, + GlslType = glslType, + BindlessArray = array.Name, + ArrayBinding = array.Value, + PushOffset = member.Offset, + Order = variant.Samplers.Count, + }); + } + } + foreach (string slot in slotTypes.Keys.OrderBy(s => s, StringComparer.Ordinal)) + { + bool inRecord = record?.Members.Any(m => m.Name == slot) ?? false; + if (inRecord) errors.Add(label + ": sampler slot '" + slot + "' is declared in the program record; slots live in the push block"); + } + + foreach (UniformMember member in FrameGlobals.Members) + { + string? owner = FrameGlobals.OwnerOf(member.Name); + if (owner != null && NativeIncludesFor(owner).Any(includes.Contains)) variant.FrameMembers.Add(member.Name); + } + + var frameTextures = new SortedDictionary(); + var storage = new SortedDictionary<(int, int), NativeStorageBinding>(); + foreach (SpirvModuleReflection module in new[] { vertex, fragment }) + { + foreach (SpirvDescriptorBinding binding in module.Bindings) + { + if (binding.Set != SetConvention.StorageSet || binding.Binding == SetConvention.ProgramRecordBinding) continue; + storage.TryAdd((binding.Set, binding.Binding), new NativeStorageBinding + { + Name = binding.Name, + Set = binding.Set, + Binding = binding.Binding, + DescriptorType = binding.Kind == SpirvDescriptorKind.StorageBuffer ? "storageBuffer" : "uniformBuffer", + ArrayLength = binding.ArrayLength, + RuntimeArray = binding.RuntimeArray, + }); + } + } + foreach (SpirvModuleReflection module in new[] { shippedVertex, shippedFragment }) + { + foreach (SpirvDescriptorBinding binding in module.Bindings) + { + if (!module.UsedVariables.Contains(binding.VariableId)) continue; + if (binding.Set == SetConvention.FrameSet) + { + SetConvention.Binding texture = Array.Find(SetConvention.FrameTextures, b => b.Value == binding.Binding); + if (texture.Name != null) + { + frameTextures[binding.Binding] = new NativeFrameTexture + { + Name = texture.Name, + GlslType = texture.GlslType, + Binding = texture.Value, + }; + } + } + else if (storage.TryGetValue((binding.Set, binding.Binding), out NativeStorageBinding? declared)) + { + declared.Used = true; + } + } + } + variant.FrameTextures.AddRange(frameTextures.Values); + variant.StorageBindings.AddRange(storage.Values); + + variant.VertexInputs = vertex.Inputs.Select(ToNative).ToList(); + variant.FragmentOutputs = fragment.Outputs.Select(ToNative).ToList(); + foreach (int location in shippedFragment.WrittenOutputLocations) + { + if (location < 32) variant.WrittenOutputs |= 1u << location; + } + + var constants = new SortedDictionary(); + foreach (SpirvModuleReflection module in new[] { vertex, fragment }) + { + foreach (SpirvSpecConstant constant in module.SpecConstants) + { + var native = new NativeSpecConstant + { + Id = constant.SpecId, + Name = constant.Name, + Type = constant.GlslType, + Default = constant.DefaultValue, + }; + if (!constants.TryGetValue(constant.SpecId, out NativeSpecConstant? existing)) + { + constants[constant.SpecId] = native; + } + else if (existing.Name != native.Name || existing.Type != native.Type || !existing.Default.Equals(native.Default)) + { + errors.Add(label + ": specialization constant " + constant.SpecId + " is '" + existing.Name + "' " + existing.Type + + " = " + existing.Default + " in one stage and '" + native.Name + "' " + native.Type + " = " + native.Default + " in the other"); + } + } + } + variant.SpecializationConstants.AddRange(constants.Values); + + return errors.Count == errorsBefore; + } + + /// The native include names that stand for a GLSL 330 owner file (fogandlight.vsh is fogandlight.vert.glsl). + internal static IEnumerable NativeIncludesFor(string owner) + { + string stem = Path.GetFileNameWithoutExtension(owner); + string extension = Path.GetExtension(owner); + string stage = extension switch + { + ".vsh" => "vert", + ".fsh" => "frag", + ".gsh" => "geom", + _ => "", + }; + if (stage.Length > 0) yield return stem + "." + stage + ".glsl"; + yield return stem + ".glsl"; + } + + private static SpirvBlock? RecordOf(SpirvModuleReflection module) => + module.Bindings.Find(b => b.Set == SetConvention.StorageSet && b.Binding == SetConvention.ProgramRecordBinding)?.Block; + + private static SpirvBlock? Agree(string label, string what, SpirvBlock? a, SpirvBlock? b, List errors) + { + if (a == null || b == null) return a ?? b; + if (Describe(a) != Describe(b)) + { + errors.Add(label + ": the " + what + " differs between the stages: " + Describe(a) + " vs " + Describe(b)); + } + return a; + } + + private static string Describe(SpirvBlock block) => + block.TypeName + "{" + string.Join(";", block.Members.Select(m => + m.GlslType + " " + m.Name + (m.ArrayLength != 0 ? "[" + m.ArrayLength + "]" : "") + "@" + m.Offset)) + "}"; + + private static void CheckConvention(string label, SpirvDescriptorBinding binding, List errors) + { + string where = label + ": '" + binding.Name + "' at set " + binding.Set + " binding " + binding.Binding + + " (reflected " + binding.Kind + " " + binding.GlslType + (binding.RuntimeArray ? "[]" : binding.ArrayLength > 0 ? "[" + binding.ArrayLength + "]" : "") + ")"; + switch (binding.Set) + { + case SetConvention.FrameSet: + if (binding.Binding == SetConvention.FrameGlobalsBinding) + { + if (binding.Kind != SpirvDescriptorKind.UniformBuffer) errors.Add(where + " must be the FrameGlobals uniform block"); + return; + } + SetConvention.Binding frame = Array.Find(SetConvention.FrameTextures, b => b.Value == binding.Binding); + if (frame.Name == null) errors.Add(where + " is not a binding of set 0 (bindings.glsl)"); + else if (binding.Kind != SpirvDescriptorKind.CombinedImageSampler || binding.GlslType != frame.GlslType || binding.ArrayLength != 0) + { + errors.Add(where + " must be " + frame.GlslType + " " + frame.Name); + } + return; + case SetConvention.TextureSet: + SetConvention.Binding array = Array.Find(SetConvention.TextureArrays, b => b.Value == binding.Binding); + if (array.Name == null) errors.Add(where + " is not a binding of set 1 (bindings.glsl)"); + // glslang sizes a `[]` array that is never indexed as a one-element array, so an + // unused bindless array in the unoptimised twin is sized, not runtime. + else if (binding.Kind != SpirvDescriptorKind.CombinedImageSampler || binding.GlslType != array.GlslType || + !(binding.RuntimeArray || binding.ArrayLength > 0)) + { + errors.Add(where + " must be " + array.GlslType + " " + array.Name + "[]"); + } + return; + case SetConvention.StorageSet: + if (binding.Binding == SetConvention.ProgramRecordBinding) + { + if (binding.Kind != SpirvDescriptorKind.UniformBuffer) errors.Add(where + " must be the program record, a uniform block"); + return; + } + if (Array.FindIndex(SetConvention.StorageBuffers, b => b.Value == binding.Binding) < 0) + { + errors.Add(where + " is not a binding of set 2 (bindings.glsl)"); + } + else if (binding.Kind != SpirvDescriptorKind.StorageBuffer) + { + errors.Add(where + " must be a storage buffer"); + } + return; + default: + errors.Add(where + ": only sets 0-" + (SetConvention.SetCount - 1) + " exist (bindings.glsl)"); + return; + } + } + + private static NativeBlock? ToNative(SpirvBlock? block) => block == null ? null : new NativeBlock + { + TypeName = block.TypeName, + Size = block.Size, + Members = block.Members.Select(m => new NativeMember + { + Name = m.Name, + Type = m.GlslType, + Offset = m.Offset, + Size = m.Size, + ArrayLength = m.ArrayLength, + }).ToList(), + }; + + private static NativeInterfaceVariable ToNative(SpirvInterfaceVariable variable) => new() + { + Location = variable.Location, + Name = variable.Name, + Type = variable.GlslType, + ArrayLength = variable.ArrayLength, + }; + + // ------------------------------------------------------------------ source scanning + + /// + /// The file's text with every #include replaced by the included file, recursively. + /// Include guards are the files' own business, as with GL_GOOGLE_include_directive. + /// + internal static string ExpandIncludes(string path, string sourceDirectory, ISet included, int depth = 0) + { + if (depth > MaxIncludeDepth) throw new InvalidDataException("includes nest deeper than " + MaxIncludeDepth + " at " + path); + string text = File.ReadAllText(path).Replace("\r\n", "\n"); + string directory = Path.GetDirectoryName(path) ?? sourceDirectory; + + return IncludeDirective.Replace(text, match => + { + string name = match.Groups[1].Value; + string? resolved = new[] + { + Path.Combine(directory, name), + Path.Combine(sourceDirectory, IncludeDirectoryName, name), + } + .FirstOrDefault(File.Exists); + if (resolved == null) + { + throw new InvalidDataException(Path.GetFileName(path) + " includes '" + name + "', found neither beside it nor in " + IncludeDirectoryName + "/"); + } + included.Add(Path.GetFileName(resolved)); + return ExpandIncludes(resolved, sourceDirectory, included, depth + 1); + }); + } + + /// The variant axes any conditional in the expanded stages tests, sorted. + internal static List AxesOf(string program, IEnumerable texts, List errors) + { + var axes = new SortedSet(StringComparer.Ordinal); + foreach (string text in texts) + { + foreach (Match match in ConditionalDirective.Matches(text)) + { + string directive = match.Groups[1].Value; + string condition = match.Groups[2].Value; + foreach (Match token in Identifier.Matches(condition)) + { + if (Array.BinarySearch(VariantAxes, token.Value, StringComparer.Ordinal) < 0) continue; + if (directive is "ifdef" or "ifndef") + { + errors.Add(program + ": #" + directive + " " + token.Value + " - every axis is always defined (0 or 1); test it with #if"); + } + axes.Add(token.Value); + } + foreach (Match defined in DefinedAxis.Matches(condition)) + { + if (Array.BinarySearch(VariantAxes, defined.Groups[1].Value, StringComparer.Ordinal) >= 0) + { + errors.Add(program + ": defined(" + defined.Groups[1].Value + ") - every axis is always defined (0 or 1); test its value"); + } + } + } + } + return axes.ToList(); + } + + /// Sampler slot name to GLSL sampler type, from every use. + internal static Dictionary SamplerSlotDeclarations(string program, IEnumerable texts, List errors) + { + var slots = new Dictionary(StringComparer.Ordinal); + foreach (string text in texts) + { + // bindings.glsl documents the macro with an example use in a comment. + foreach (Match match in SamplerSlot.Matches(StripComments(text))) + { + // The macro's own #define line is excluded by the pattern; a use can still share a + // line with other text, so take every use on the matched line. + foreach (Match use in SamplerSlotAnywhere.Matches(match.Value)) + { + string type = use.Groups[1].Value; + string name = use.Groups[2].Value; + if (Array.FindIndex(SetConvention.TextureArrays, b => b.GlslType == type) < 0) + { + errors.Add(program + ": sampler slot '" + name + "' has type " + type + ", which has no bindless array (bindings.glsl)"); + continue; + } + if (slots.TryGetValue(name, out string? existing) && existing != type) + { + errors.Add(program + ": sampler slot '" + name + "' is declared both " + existing + " and " + type); + continue; + } + slots[name] = type; + } + } + } + return slots; + } + + /// The text with // and /* */ comments blanked out, line breaks kept. + internal static string StripComments(string text) + { + var builder = new StringBuilder(text.Length); + int i = 0; + while (i < text.Length) + { + if (text[i] == '/' && i + 1 < text.Length && text[i + 1] == '/') + { + while (i < text.Length && text[i] != '\n') i++; + } + else if (text[i] == '/' && i + 1 < text.Length && text[i + 1] == '*') + { + i += 2; + while (i < text.Length && !(text[i] == '*' && i + 1 < text.Length && text[i + 1] == '/')) + { + if (text[i] == '\n') builder.Append('\n'); + i++; + } + i = Math.Min(i + 2, text.Length); + builder.Append(' '); + } + else + { + builder.Append(text[i]); + i++; + } + } + return builder.ToString(); + } + + // ------------------------------------------------------------------ output + + /// + /// Writes a full build to <outputRoot>/shaders-vk: every SPIR-V file and the + /// manifest, rewriting only files whose bytes changed, and deleting SPIR-V the build no longer + /// produces. + /// + public static void Write(NativeShaderBuildResult result, string outputRoot) + { + string directory = Path.Combine(outputRoot, NativeShaderManifest.DirectoryName); + Directory.CreateDirectory(directory); + foreach (string stale in Directory.GetFiles(directory, "*.spv")) + { + if (!result.Files.ContainsKey(Path.GetFileName(stale))) File.Delete(stale); + } + foreach ((string name, byte[] bytes) in result.Files) WriteIfChanged(Path.Combine(directory, name), bytes); + WriteIfChanged(Path.Combine(directory, NativeShaderManifest.FileName), Encoding.UTF8.GetBytes(result.Manifest.ToJson())); + } + + /// + /// Merges a one-program build into the manifest already in , which + /// must have been written by the same toolchain; replaces that program's SPIR-V. + /// + public static void WriteSingle(NativeShaderBuildResult result, string outputRoot) + { + string directory = Path.Combine(outputRoot, NativeShaderManifest.DirectoryName); + string manifestPath = Path.Combine(directory, NativeShaderManifest.FileName); + if (!File.Exists(manifestPath)) throw new InvalidDataException("no manifest at " + manifestPath + "; run --build first"); + + NativeShaderManifest existing = NativeShaderManifest.Load(manifestPath); + if (existing.Toolchain != result.Manifest.Toolchain) + { + throw new InvalidDataException("the manifest at " + manifestPath + " was built by another toolchain; run --build"); + } + + foreach (NativeProgram program in result.Manifest.Programs) + { + NativeProgram? old = existing.FindProgram(program.Name); + if (old != null) + { + foreach (NativeStage stage in old.Variants.SelectMany(v => v.Stages)) + { + string stalePath = Path.Combine(directory, stage.Spirv); + if (!result.Files.ContainsKey(stage.Spirv) && File.Exists(stalePath)) File.Delete(stalePath); + } + existing.Programs.Remove(old); + } + existing.Programs.Add(program); + } + existing.Programs.Sort((a, b) => string.CompareOrdinal(a.Name, b.Name)); + + foreach ((string name, byte[] bytes) in result.Files) WriteIfChanged(Path.Combine(directory, name), bytes); + WriteIfChanged(manifestPath, Encoding.UTF8.GetBytes(existing.ToJson())); + } + + /// Every way the files in <outputRoot>/shaders-vk differ from a fresh build; empty when identical. + public static List Compare(NativeShaderBuildResult result, string outputRoot) + { + var differences = new List(); + string directory = Path.Combine(outputRoot, NativeShaderManifest.DirectoryName); + string manifestPath = Path.Combine(directory, NativeShaderManifest.FileName); + + if (!File.Exists(manifestPath)) + { + differences.Add("missing " + manifestPath); + } + else if (File.ReadAllText(manifestPath) != result.Manifest.ToJson()) + { + differences.Add(NativeShaderManifest.FileName + " differs from a fresh build"); + } + + foreach ((string name, byte[] bytes) in result.Files) + { + string path = Path.Combine(directory, name); + if (!File.Exists(path)) differences.Add("missing " + name); + else if (!File.ReadAllBytes(path).AsSpan().SequenceEqual(bytes)) differences.Add(name + " differs (sha256 " + + Convert.ToHexStringLower(SHA256.HashData(File.ReadAllBytes(path))) + ", fresh build " + Convert.ToHexStringLower(SHA256.HashData(bytes)) + ")"); + } + + if (Directory.Exists(directory)) + { + foreach (string file in Directory.GetFiles(directory, "*.spv")) + { + if (!result.Files.ContainsKey(Path.GetFileName(file))) differences.Add("unexpected " + Path.GetFileName(file)); + } + } + return differences; + } + + private static void WriteIfChanged(string path, byte[] bytes) + { + if (File.Exists(path) && File.ReadAllBytes(path).AsSpan().SequenceEqual(bytes)) return; + File.WriteAllBytes(path, bytes); + } +} + +/// +/// The command line of tools/shader-compiler, kept here so tests drive exactly what the +/// build runs: +/// --build <src> <out>, --verify <src> <out>, +/// --single <program> <src> <out>. Output lands in <out>/shaders-vk. +/// Exit codes: 0 success, 1 build or verify failure, 2 usage. +/// +internal static class NativeShaderTool +{ + public const string Usage = + "usage: Optimum.Shaders.Compiler --build \n" + + " Optimum.Shaders.Compiler --verify \n" + + " Optimum.Shaders.Compiler --single "; + + public static int Run(string[] args, TextWriter output, TextWriter error) + { + string mode = args.Length > 0 ? args[0] : ""; + int expected = mode == "--single" ? 4 : 3; + if (mode is not ("--build" or "--verify" or "--single") || args.Length != expected) + { + error.WriteLine(Usage); + return 2; + } + + string? program = mode == "--single" ? args[1] : null; + string source = args[expected - 2]; + string outputRoot = args[expected - 1]; + + using var compiler = new ShaderCompiler(); + NativeShaderBuildResult result = new NativeShaderBuilder(compiler).Build(source, program); + if (!result.Success) + { + foreach (string message in result.Errors) error.WriteLine("error: " + message); + error.WriteLine("shader-compiler: " + result.Errors.Count + " error(s); nothing written"); + return 1; + } + + int variants = result.Manifest.Programs.Sum(p => p.Variants.Count); + string summary = result.Manifest.Programs.Count + " program(s), " + variants + " variant(s), " + result.Files.Count + " SPIR-V file(s)"; + try + { + switch (mode) + { + case "--build": + NativeShaderBuilder.Write(result, outputRoot); + output.WriteLine("shader-compiler: built " + summary + " into " + Path.Combine(outputRoot, NativeShaderManifest.DirectoryName)); + return 0; + case "--single": + NativeShaderBuilder.WriteSingle(result, outputRoot); + output.WriteLine("shader-compiler: rebuilt " + program + ", " + summary); + return 0; + default: + List differences = NativeShaderBuilder.Compare(result, outputRoot); + foreach (string difference in differences) error.WriteLine("differs: " + difference); + if (differences.Count > 0) + { + error.WriteLine("shader-compiler: " + differences.Count + " difference(s) against a fresh build of " + source); + return 1; + } + output.WriteLine("shader-compiler: verified " + summary); + return 0; + } + } + catch (Exception failure) when (failure is IOException or InvalidDataException or UnauthorizedAccessException) + { + error.WriteLine("error: " + failure.Message); + return 1; + } + } +} diff --git a/Optimum.Render.Vulkan/Shaders/NativeShaderLibrary.cs b/Optimum.Render.Vulkan/Shaders/NativeShaderLibrary.cs new file mode 100644 index 00000000..03c90dac --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/NativeShaderLibrary.cs @@ -0,0 +1,526 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.IO; +using System.Security.Cryptography; +using System.Text.RegularExpressions; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// The specialization constants of one native program, as a pipeline's stages take them. +internal sealed class NativeSpecialization +{ + /// One constant: its id, and where its 4 bytes sit in . + public readonly record struct Entry(uint Id, uint Offset, uint Size); + + public Entry[] Entries = Array.Empty(); + public byte[] Data = Array.Empty(); +} + +/// What the GLSL 330 text of a program says that the manifest does not: sampler units and initializers. +internal sealed class GlslUniformOracle +{ + /// + /// ShaderProgram.collectUniformNames (build/VintagestoryLib/Vintagestory.Client.NoObf/ShaderProgram.cs:57-68), + /// pattern and options verbatim. + /// + private static readonly Regex CollectUniformNames = new( + "(\\s|\\r\\n)uniform\\s*(?float|int|ivec2|ivec3|ivec4|vec2|vec3|vec4|sampler2DShadow|sampler2D|samplerCube|mat3|mat4x3|mat4)\\s*(\\[[\\d\\w]+\\])?\\s*(?[\\d\\w]+)", + RegexOptions.IgnoreCase | RegexOptions.ExplicitCapture | RegexOptions.Compiled); + + private static readonly Regex Initializer = new( + @"\buniform\s+\w+\s+(?\w+)\s*=\s*(?[^;]+);", RegexOptions.Compiled); + + private static readonly Regex BlockComment = new(@"/\*.*?\*/", RegexOptions.Singleline | RegexOptions.Compiled); + private static readonly Regex LineComment = new(@"//[^\n]*", RegexOptions.Compiled); + + private readonly List<(string Name, string Type, int Unit)> _samplers = new(); + private readonly Dictionary _initializers = new(StringComparer.Ordinal); + + /// The include-expanded GLSL 330 stage texts in the order the client scans them (vertex, fragment). + public GlslUniformOracle(IEnumerable stageCodes) + { + var units = new Dictionary(StringComparer.Ordinal); + var order = new List<(string Name, string Type)>(); + foreach (string code in stageCodes) + { + foreach (Match match in CollectUniformNames.Matches(code)) + { + string type = match.Groups["type"].Value; + if (!type.Contains("sampler", StringComparison.Ordinal)) continue; + string name = match.Groups["var"].Value; + // textureLocations[value] = textureLocations.Count: a name seen again takes the current count. + units[name] = units.Count; + order.Add((name, type)); + } + + string stripped = LineComment.Replace(BlockComment.Replace(code, " "), " "); + foreach (Match match in Initializer.Matches(stripped)) + { + _initializers.TryAdd(match.Groups["name"].Value, match.Groups["value"].Value.Trim()); + } + } + + var seen = new HashSet(StringComparer.Ordinal); + foreach ((string name, string type) in order) + { + if (seen.Add(name)) _samplers.Add((name, type, units[name])); + } + } + + /// Each sampler name once, with the unit the client assigns it, in unit order. + public IEnumerable<(string Name, string Type, int Unit)> SamplersByUnit() + { + var sorted = new List<(string Name, string Type, int Unit)>(_samplers); + sorted.Sort((a, b) => a.Unit.CompareTo(b.Unit)); + return sorted; + } + + /// The initializer a GLSL 330 declaration of carries, or null. + public string? InitializerOf(string name) => _initializers.TryGetValue(name, out string? value) ? value : null; +} + +/// +/// The native shaders at runtime (docs/vulkan.md): the manifest and SPIR-V +/// beside Optimum.Render.Vulkan.dll, or a source tree compiled at device start, looked up per +/// program at VulkanDevice.LinkProgram. +/// +/// Loaded once. SPIR-V is read and its SHA-256 checked the first time a variant is linked; a file that +/// fails the check fails that variant for the life of the library, and the program links through the +/// rewriter. +/// +internal sealed class NativeShaderLibrary +{ + /// + /// 0 forces the rewriter for every program (A/B runs); links natively even + /// where the launcher's mod shader scan says a mod replaced the program (development runs without the launcher). + /// + public const string EnabledVariable = "OPTIMUM_VK_NATIVE_SHADERS"; + + /// The value that ignores the mod shader scan. + public const string ForceValue = "force"; + + /// Whether 's value asks to ignore the mod shader scan. + public static bool IgnoresModScan(string? enabledVariable) => + string.Equals(enabledVariable?.Trim(), ForceValue, StringComparison.OrdinalIgnoreCase); + + /// A sources/shaders-vk tree to compile at device start instead of the shipped manifest. + public const string SourceVariable = "OPTIMUM_VK_SHADER_SOURCE"; + + public enum Mode { Off, Directory, Source } + + /// How a link request came out. + public enum Outcome + { + /// No native program of that name (or a geometry stage): the rewriter, as before. + Miss, + /// Linked from the manifest. + Native, + /// A native program exists but could not be used: the rewriter, counted as failed. + Failed, + } + + public NativeShaderManifest Manifest { get; } + + /// Where the manifest came from, for the log. + public string Origin { get; } + + private readonly string? _directory; + private readonly IReadOnlyDictionary? _files; + private readonly Dictionary _verified = new(StringComparer.Ordinal); + private readonly Dictionary _rejected = new(StringComparer.Ordinal); + private readonly object _lock = new(); + + private NativeShaderLibrary(NativeShaderManifest manifest, string origin, string? directory, IReadOnlyDictionary? files) + { + Manifest = manifest; + Origin = origin; + _directory = directory; + _files = files; + } + + // ------------------------------------------------------------------ loading + + /// + /// Decides where native shaders come from. Off wins (the environment's 0 or the device setting); + /// then an explicit directory (tests), then , then shaders-vk beside + /// the renderer assembly. + /// + public static (Mode Mode, string? Path, string Reason) Resolve( + bool? enabledSetting, string? directorySetting, string? enabledVariable, string? sourceVariable, string? assemblyDirectory) + { + if (enabledSetting == false) return (Mode.Off, null, "native shaders off by device setting"); + if (enabledVariable?.Trim() == "0") return (Mode.Off, null, EnabledVariable + "=0"); + if (!string.IsNullOrEmpty(directorySetting)) return (Mode.Directory, directorySetting, ""); + if (!string.IsNullOrWhiteSpace(sourceVariable)) return (Mode.Source, sourceVariable, ""); + if (string.IsNullOrEmpty(assemblyDirectory)) return (Mode.Off, null, "renderer assembly location unknown"); + return (Mode.Directory, System.IO.Path.Combine(assemblyDirectory, NativeShaderManifest.DirectoryName), ""); + } + + /// + /// Reads shaders.manifest.json from . Null, with the one reason, when it is + /// missing, malformed, of another schema version, or built with other compile options than . + /// + /// A toolchain is ShaderCompiler.Identity: the options, then after the last ; the shaderc library + /// (its SHA-256, or the Silk package). Only the options must match: a manifest built on another platform or with + /// another shaderc build is still accepted, because every stage's SHA-256 pins the SPIR-V bytes themselves. The + /// differing library is then the non-empty of a loaded library, for the status line. + /// + public static NativeShaderLibrary? Load(string directory, string toolchain, out string reason) + { + string path = System.IO.Path.Combine(directory, NativeShaderManifest.FileName); + if (!File.Exists(path)) + { + reason = "no manifest at " + path; + return null; + } + + NativeShaderManifest manifest; + try + { + manifest = NativeShaderManifest.Load(path); + } + catch (Exception error) when (error is InvalidDataException or IOException or UnauthorizedAccessException + or InvalidOperationException or FormatException or System.Text.Json.JsonException) + { + reason = "manifest " + path + " rejected: " + error.Message; + return null; + } + + reason = ""; + if (manifest.Toolchain != toolchain) + { + (string builtOptions, string builtLibrary) = SplitToolchain(manifest.Toolchain); + (string ownOptions, string ownLibrary) = SplitToolchain(toolchain); + if (builtOptions != ownOptions) + { + reason = "manifest " + path + " was built by toolchain '" + manifest.Toolchain + "', this renderer compiles with '" + toolchain + "'"; + return null; + } + reason = "manifest built with shader library '" + builtLibrary + "', this renderer has '" + ownLibrary + + "' (same options; the SPIR-V is pinned by its per-stage sha256)"; + } + + return new NativeShaderLibrary(manifest, path, directory, null); + } + + /// A toolchain identity split at its last ;: the compile options and the shaderc library. + internal static (string Options, string Library) SplitToolchain(string? toolchain) + { + toolchain ??= ""; + int split = toolchain.LastIndexOf(';'); + return split < 0 ? (toolchain, "") : (toolchain[..split], toolchain[(split + 1)..]); + } + + /// + /// Compiles a source tree through , the offline tool's library. Programs + /// that built are usable even when others failed; then names the failures. + /// + public static NativeShaderLibrary? BuildFromSource(string sourceDirectory, ShaderCompiler compiler, out string reason) + { + if (!Directory.Exists(sourceDirectory)) + { + reason = "source tree " + sourceDirectory + " not found"; + return null; + } + + NativeShaderBuildResult result; + try + { + result = new NativeShaderBuilder(compiler).Build(sourceDirectory); + } + catch (Exception error) when (error is IOException or UnauthorizedAccessException or InvalidDataException) + { + reason = "source tree " + sourceDirectory + " did not build: " + error.Message; + return null; + } + + reason = result.Success + ? "" + : result.Errors.Count + " build error(s) in " + sourceDirectory + ", first: " + result.Errors[0]; + return new NativeShaderLibrary(result.Manifest, "source " + sourceDirectory, null, result.Files); + } + + /// A library over an in-memory build, for tests. + internal static NativeShaderLibrary FromBuild(NativeShaderBuildResult result) => + new(result.Manifest, "in-memory build", null, result.Files); + + /// The verified bytes of one stage's SPIR-V; false with the reason when unreadable or its hash disagrees. + public bool TryGetSpirv(NativeStage stage, out byte[] spirv, out string error) + { + lock (_lock) + { + if (_verified.TryGetValue(stage.Spirv, out spirv!)) + { + error = ""; + return true; + } + if (_rejected.TryGetValue(stage.Spirv, out error!)) + { + spirv = Array.Empty(); + return false; + } + + byte[]? bytes = null; + if (_files != null) + { + if (!_files.TryGetValue(stage.Spirv, out bytes)) error = stage.Spirv + " is not in the build"; + } + else + { + string path = System.IO.Path.Combine(_directory!, stage.Spirv); + try + { + bytes = File.ReadAllBytes(path); + } + catch (Exception readError) when (readError is IOException or UnauthorizedAccessException) + { + error = stage.Spirv + " unreadable: " + readError.Message; + } + } + + if (bytes != null) + { + string actual = Convert.ToHexStringLower(SHA256.HashData(bytes)); + if (actual == stage.Sha256 && bytes.Length % 4 == 0) + { + _verified[stage.Spirv] = bytes; + spirv = bytes; + error = ""; + return true; + } + error = stage.Spirv + " sha256 " + actual + ", manifest says " + stage.Sha256; + } + + _rejected[stage.Spirv] = error; + spirv = Array.Empty(); + return false; + } + } + + // ------------------------------------------------------------------ defines + + private static readonly Regex Define = new(@"^[ \t]*#[ \t]*define[ \t]+(\w+)(?:[ \t]+([^\r\n]*?))?[ \t]*\r?$", RegexOptions.Multiline | RegexOptions.Compiled); + + /// + /// The #defines of a program's prefixes (ShaderRegistry.registerDefaultShaderCodePrefixes plus the + /// program's own), stages merged in the order given. A define without a value maps to the empty string. + /// + public static Dictionary ParseDefines(IEnumerable prefixes) + { + var defines = new Dictionary(StringComparer.Ordinal); + foreach (string prefix in prefixes) + { + foreach (Match match in Define.Matches(prefix ?? "")) + { + defines[match.Groups[1].Value] = match.Groups[2].Value.Trim(); + } + } + return defines; + } + + /// The per-registration axes the GLSL 330 sources test with defined(): present is 1. + private static readonly HashSet PresenceAxes = new(StringComparer.Ordinal) { "ALLOWDEPTHOFFSET", "GLOWSUB", "VEC3SCALE" }; + + /// + /// The value of each axis of as the manifest keys it (contract section 5): + /// GBUFFER is SSAOLEVEL > 0, the per-registration axes are 1 when defined, every other axis is + /// its define's value. An absent define is 0, as in an #if. False for a value that is not 0 or 1. + /// + public static bool TryAxisValues(IEnumerable axes, IReadOnlyDictionary defines, + out Dictionary values, out string error) + { + values = new Dictionary(StringComparer.Ordinal); + foreach (string axis in axes) + { + int value; + if (PresenceAxes.Contains(axis)) + { + value = defines.ContainsKey(axis) ? 1 : 0; + } + else if (axis == "GBUFFER") + { + if (!TryIntDefine(defines, "SSAOLEVEL", out int ssao, out error)) return false; + value = ssao > 0 ? 1 : 0; + } + else + { + if (!TryIntDefine(defines, axis, out value, out error)) return false; + if (value is not (0 or 1)) + { + error = "axis " + axis + " is " + value + ", the manifest has 0 and 1"; + return false; + } + } + values[axis] = value; + } + error = ""; + return true; + } + + private static bool TryIntDefine(IReadOnlyDictionary defines, string name, out int value, out string error) + { + error = ""; + if (!defines.TryGetValue(name, out string? text)) + { + value = 0; + return true; + } + if (int.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out value)) return true; + error = "#define " + name + " '" + text + "' is not an integer"; + return false; + } + + /// The manifest variant key of for a program's defines; null with the reason when it has none. + public static string? VariantKeyFor(NativeProgram program, IReadOnlyDictionary defines, out string error) + { + if (!TryAxisValues(program.Axes, defines, out Dictionary values, out error)) return null; + return NativeShaderManifest.VariantKey(program.Axes, values); + } + + /// + /// The specialization data for a variant: every constant the manifest lists, valued from the define + /// maps it to (the same setting ShaderRegistry stamps into the prefix), + /// 0 when the prefix does not define it. + /// + public static bool TryBuildSpecialization(NativeVariant variant, IReadOnlyDictionary defines, + out NativeSpecialization specialization, out string error) + { + var entries = new List(); + var data = new List(); + foreach (NativeSpecConstant constant in variant.SpecializationConstants) + { + SpecializationConvention.Constant? convention = null; + foreach (SpecializationConvention.Constant candidate in SpecializationConvention.Constants) + { + if (candidate.Id == (uint)constant.Id) convention = candidate; + } + if (convention == null || convention.Value.Name != constant.Name) + { + specialization = new NativeSpecialization(); + error = "specialization constant " + constant.Id + " '" + constant.Name + "' is not in the convention"; + return false; + } + + defines.TryGetValue(convention.Value.Define, out string? text); + byte[] bytes; + switch (constant.Type) + { + case "float": + float f = 0; + if (text != null && !float.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out f)) + { + specialization = new NativeSpecialization(); + error = "#define " + convention.Value.Define + " '" + text + "' is not a float"; + return false; + } + bytes = BitConverter.GetBytes(f); + break; + case "int" or "uint" or "bool": + int i = 0; + if (text != null && !int.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out i)) + { + specialization = new NativeSpecialization(); + error = "#define " + convention.Value.Define + " '" + text + "' is not an integer"; + return false; + } + bytes = BitConverter.GetBytes(constant.Type == "bool" ? (i != 0 ? 1 : 0) : i); + break; + default: + specialization = new NativeSpecialization(); + error = "specialization constant '" + constant.Name + "' has type " + constant.Type; + return false; + } + entries.Add(new NativeSpecialization.Entry((uint)constant.Id, (uint)data.Count, (uint)bytes.Length)); + data.AddRange(bytes); + } + + specialization = new NativeSpecialization { Entries = entries.ToArray(), Data = data.ToArray() }; + error = ""; + return true; + } + + // ------------------------------------------------------------------ linking + + /// + /// Looks up and, on a hit, builds the program from the manifest: SPIR-V per stage, + /// the layout, the specialization. are the GLSL 330 stages the program carries, for + /// the defines, the sampler units and the initializers. + /// + public Outcome TryLink(string passName, IReadOnlyList stages, + out TranslatedProgram? program, out string detail) + { + program = null; + detail = ""; + + ShaderStageSource? vertex = null, fragment = null; + foreach (ShaderStageSource stage in stages) + { + if (stage.Stage == EnumShaderType.VertexShader) vertex = stage; + else if (stage.Stage == EnumShaderType.FragmentShader) fragment = stage; + else return Outcome.Miss; + } + + NativeProgram? native = Manifest.FindProgram(passName); + if (native == null) return Outcome.Miss; + + if (vertex == null || fragment == null) + { + detail = "the GLSL 330 program lacks a vertex or fragment stage"; + return Outcome.Failed; + } + + Dictionary defines = ParseDefines(new[] { vertex.PrefixCode, fragment.PrefixCode }); + string? key = VariantKeyFor(native, defines, out string keyError); + if (key == null) + { + detail = keyError; + return Outcome.Failed; + } + detail = key; + + NativeVariant? variant = native.Variants.Find(v => v.Key == key); + if (variant == null) + { + detail = "no variant '" + key + "'"; + return Outcome.Failed; + } + + var translated = new TranslatedProgram { IsNative = true }; + foreach ((string stageName, EnumShaderType type) in new[] + { ("vertex", EnumShaderType.VertexShader), ("fragment", EnumShaderType.FragmentShader) }) + { + NativeStage? stage = variant.Stages.Find(s => s.Stage == stageName); + if (stage == null) + { + detail = "[" + key + "] has no " + stageName + " stage"; + return Outcome.Failed; + } + if (!TryGetSpirv(stage, out byte[] spirv, out string spirvError)) + { + detail = "[" + key + "] " + spirvError; + return Outcome.Failed; + } + translated.Spirv[type] = spirv; + } + + if (!TryBuildSpecialization(variant, defines, out NativeSpecialization specialization, out string specError)) + { + detail = "[" + key + "] " + specError; + return Outcome.Failed; + } + translated.Specialization = specialization; + + var oracle = new GlslUniformOracle(new[] { vertex.Code, fragment.Code }); + translated.Layout = ProgramInterfaceLayout.FromNative(variant, oracle); + if (translated.Layout.HasErrors) + { + detail = "[" + key + "] " + string.Join("; ", translated.Layout.Errors); + return Outcome.Failed; + } + + program = translated; + return Outcome.Native; + } +} diff --git a/Optimum.Render.Vulkan/Shaders/NativeShaderManifest.cs b/Optimum.Render.Vulkan/Shaders/NativeShaderManifest.cs new file mode 100644 index 00000000..6eeb1e21 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/NativeShaderManifest.cs @@ -0,0 +1,516 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.IO; +using System.Text; +using System.Text.Json; + +namespace Optimum.Render.Vulkan.Shaders; + +/// +/// shaders.manifest.json: what the offline compiler (tools/shader-compiler) produced +/// from sources/shaders-vk and what the runtime links against +/// (docs/vulkan.md). +/// +/// Written and read with and directly: +/// the output is byte-for-byte deterministic (the --verify gate compares it as bytes), and +/// no reflection-based serializer is involved at runtime. +/// +internal sealed class NativeShaderManifest +{ + /// Bumped whenever a field is added, removed or changes meaning; readers refuse any other value. + public const int CurrentSchemaVersion = 1; + + public const string FileName = "shaders.manifest.json"; + + /// The output directory, beside Optimum.Render.Vulkan.dll; never under assets/. + public const string DirectoryName = "shaders-vk"; + + public int SchemaVersion = CurrentSchemaVersion; + + /// of the compiler that produced every blob. + public string Toolchain = ""; + + /// Sorted by name. + public List Programs = new(); + + public NativeProgram? FindProgram(string name) => Programs.Find(p => p.Name == name); + + public NativeVariant? Find(string program, string variantKey) => + FindProgram(program)?.Variants.Find(v => v.Key == variantKey); + + /// + /// The variant key for a set of axis values: the sorted NAME=value list, comma separated, + /// restricted to (the axes the program branches on). Empty when none. + /// + public static string VariantKey(IEnumerable axes, IReadOnlyDictionary values) + { + var names = new List(axes); + names.Sort(StringComparer.Ordinal); + var parts = new List(names.Count); + foreach (string name in names) + { + parts.Add(name + "=" + (values.TryGetValue(name, out int value) ? value : 0).ToString(CultureInfo.InvariantCulture)); + } + return string.Join(",", parts); + } + + // ------------------------------------------------------------------ writer + + public string ToJson() + { + using var stream = new MemoryStream(); + using (var writer = new Utf8JsonWriter(stream, new JsonWriterOptions { Indented = true, NewLine = "\n" })) + { + writer.WriteStartObject(); + writer.WriteNumber("schemaVersion", SchemaVersion); + writer.WriteString("toolchain", Toolchain); + writer.WriteStartArray("programs"); + foreach (NativeProgram program in Programs) WriteProgram(writer, program); + writer.WriteEndArray(); + writer.WriteEndObject(); + } + return Encoding.UTF8.GetString(stream.ToArray()) + "\n"; + } + + private static void WriteProgram(Utf8JsonWriter writer, NativeProgram program) + { + writer.WriteStartObject(); + writer.WriteString("name", program.Name); + WriteStrings(writer, "axes", program.Axes); + writer.WriteStartArray("variants"); + foreach (NativeVariant variant in program.Variants) + { + writer.WriteStartObject(); + writer.WriteString("key", variant.Key); + + writer.WriteStartArray("stages"); + foreach (NativeStage stage in variant.Stages) + { + writer.WriteStartObject(); + writer.WriteString("stage", stage.Stage); + writer.WriteString("source", stage.Source); + writer.WriteString("spirv", stage.Spirv); + writer.WriteString("sha256", stage.Sha256); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + + WriteBlock(writer, "push", variant.Push); + WriteBlock(writer, "record", variant.Record); + WriteStrings(writer, "frameMembers", variant.FrameMembers); + + writer.WriteStartArray("samplers"); + foreach (NativeSampler sampler in variant.Samplers) + { + writer.WriteStartObject(); + writer.WriteString("name", sampler.Name); + writer.WriteString("glslType", sampler.GlslType); + writer.WriteString("bindlessArray", sampler.BindlessArray); + writer.WriteNumber("arrayBinding", sampler.ArrayBinding); + writer.WriteNumber("pushOffset", sampler.PushOffset); + writer.WriteNumber("order", sampler.Order); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + + writer.WriteStartArray("frameTextures"); + foreach (NativeFrameTexture texture in variant.FrameTextures) + { + writer.WriteStartObject(); + writer.WriteString("name", texture.Name); + writer.WriteString("glslType", texture.GlslType); + writer.WriteNumber("binding", texture.Binding); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + + writer.WriteStartArray("storageBindings"); + foreach (NativeStorageBinding binding in variant.StorageBindings) + { + writer.WriteStartObject(); + writer.WriteString("name", binding.Name); + writer.WriteNumber("set", binding.Set); + writer.WriteNumber("binding", binding.Binding); + writer.WriteString("descriptorType", binding.DescriptorType); + writer.WriteNumber("arrayLength", binding.ArrayLength); + writer.WriteBoolean("runtimeArray", binding.RuntimeArray); + writer.WriteBoolean("used", binding.Used); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + + WriteInterface(writer, "vertexInputs", variant.VertexInputs); + WriteInterface(writer, "fragmentOutputs", variant.FragmentOutputs); + writer.WriteNumber("writtenOutputs", variant.WrittenOutputs); + + writer.WriteStartArray("specializationConstants"); + foreach (NativeSpecConstant constant in variant.SpecializationConstants) + { + writer.WriteStartObject(); + writer.WriteNumber("id", constant.Id); + writer.WriteString("name", constant.Name); + writer.WriteString("type", constant.Type); + if (constant.Type == "bool") writer.WriteBoolean("default", constant.Default != 0); + else writer.WriteNumber("default", constant.Default); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + + writer.WriteEndObject(); + } + writer.WriteEndArray(); + writer.WriteEndObject(); + } + + private static void WriteBlock(Utf8JsonWriter writer, string property, NativeBlock? block) + { + if (block == null) + { + writer.WriteNull(property); + return; + } + writer.WriteStartObject(property); + writer.WriteString("typeName", block.TypeName); + writer.WriteNumber("size", block.Size); + writer.WriteStartArray("members"); + foreach (NativeMember member in block.Members) + { + writer.WriteStartObject(); + writer.WriteString("name", member.Name); + writer.WriteString("type", member.Type); + writer.WriteNumber("offset", member.Offset); + writer.WriteNumber("size", member.Size); + writer.WriteNumber("arrayLength", member.ArrayLength); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + writer.WriteEndObject(); + } + + private static void WriteInterface(Utf8JsonWriter writer, string property, List variables) + { + writer.WriteStartArray(property); + foreach (NativeInterfaceVariable variable in variables) + { + writer.WriteStartObject(); + writer.WriteNumber("location", variable.Location); + writer.WriteString("name", variable.Name); + writer.WriteString("type", variable.Type); + writer.WriteNumber("arrayLength", variable.ArrayLength); + writer.WriteEndObject(); + } + writer.WriteEndArray(); + } + + private static void WriteStrings(Utf8JsonWriter writer, string property, List values) + { + writer.WriteStartArray(property); + foreach (string value in values) writer.WriteStringValue(value); + writer.WriteEndArray(); + } + + // ------------------------------------------------------------------ reader + + /// + /// Parses a manifest. Throws for malformed JSON, a missing + /// field, or a schema version other than . + /// + public static NativeShaderManifest Parse(string json) + { + JsonDocument document; + try + { + document = JsonDocument.Parse(json); + } + catch (JsonException error) + { + throw new InvalidDataException("shader manifest is not valid JSON: " + error.Message, error); + } + + using (document) + { + JsonElement root = document.RootElement; + int version = Int(root, "schemaVersion"); + if (version != CurrentSchemaVersion) + { + throw new InvalidDataException( + "shader manifest schema version " + version + ", this build reads " + CurrentSchemaVersion); + } + + var manifest = new NativeShaderManifest + { + SchemaVersion = version, + Toolchain = Str(root, "toolchain"), + }; + foreach (JsonElement program in Arr(root, "programs")) + { + manifest.Programs.Add(ReadProgram(program)); + } + return manifest; + } + } + + public static NativeShaderManifest Load(string path) => Parse(File.ReadAllText(path)); + + private static NativeProgram ReadProgram(JsonElement element) + { + var program = new NativeProgram { Name = Str(element, "name"), Axes = Strings(element, "axes") }; + foreach (JsonElement v in Arr(element, "variants")) + { + var variant = new NativeVariant + { + Key = Str(v, "key"), + Push = ReadBlock(v, "push"), + Record = ReadBlock(v, "record"), + FrameMembers = Strings(v, "frameMembers"), + VertexInputs = ReadInterface(v, "vertexInputs"), + FragmentOutputs = ReadInterface(v, "fragmentOutputs"), + WrittenOutputs = Get(v, "writtenOutputs").GetUInt32(), + }; + foreach (JsonElement s in Arr(v, "stages")) + { + variant.Stages.Add(new NativeStage + { + Stage = Str(s, "stage"), + Source = Str(s, "source"), + Spirv = Str(s, "spirv"), + Sha256 = Str(s, "sha256"), + }); + } + foreach (JsonElement s in Arr(v, "samplers")) + { + variant.Samplers.Add(new NativeSampler + { + Name = Str(s, "name"), + GlslType = Str(s, "glslType"), + BindlessArray = Str(s, "bindlessArray"), + ArrayBinding = Int(s, "arrayBinding"), + PushOffset = Int(s, "pushOffset"), + Order = Int(s, "order"), + }); + } + foreach (JsonElement t in Arr(v, "frameTextures")) + { + variant.FrameTextures.Add(new NativeFrameTexture + { + Name = Str(t, "name"), + GlslType = Str(t, "glslType"), + Binding = Int(t, "binding"), + }); + } + foreach (JsonElement b in Arr(v, "storageBindings")) + { + variant.StorageBindings.Add(new NativeStorageBinding + { + Name = Str(b, "name"), + Set = Int(b, "set"), + Binding = Int(b, "binding"), + DescriptorType = Str(b, "descriptorType"), + ArrayLength = Int(b, "arrayLength"), + RuntimeArray = Get(b, "runtimeArray").GetBoolean(), + Used = Get(b, "used").GetBoolean(), + }); + } + foreach (JsonElement c in Arr(v, "specializationConstants")) + { + string type = Str(c, "type"); + JsonElement value = Get(c, "default"); + variant.SpecializationConstants.Add(new NativeSpecConstant + { + Id = Int(c, "id"), + Name = Str(c, "name"), + Type = type, + Default = value.ValueKind switch + { + JsonValueKind.True => 1, + JsonValueKind.False => 0, + _ => value.GetDouble(), + }, + }); + } + program.Variants.Add(variant); + } + return program; + } + + private static NativeBlock? ReadBlock(JsonElement parent, string property) + { + JsonElement element = Get(parent, property); + if (element.ValueKind == JsonValueKind.Null) return null; + var block = new NativeBlock { TypeName = Str(element, "typeName"), Size = Int(element, "size") }; + foreach (JsonElement m in Arr(element, "members")) + { + block.Members.Add(new NativeMember + { + Name = Str(m, "name"), + Type = Str(m, "type"), + Offset = Int(m, "offset"), + Size = Int(m, "size"), + ArrayLength = Int(m, "arrayLength"), + }); + } + return block; + } + + private static List ReadInterface(JsonElement parent, string property) + { + var list = new List(); + foreach (JsonElement e in Arr(parent, property)) + { + list.Add(new NativeInterfaceVariable + { + Location = Int(e, "location"), + Name = Str(e, "name"), + Type = Str(e, "type"), + ArrayLength = Int(e, "arrayLength"), + }); + } + return list; + } + + private static JsonElement Get(JsonElement parent, string property) + { + if (parent.ValueKind != JsonValueKind.Object || !parent.TryGetProperty(property, out JsonElement value)) + { + throw new InvalidDataException("shader manifest: missing '" + property + "'"); + } + return value; + } + + private static string Str(JsonElement parent, string property) => + Get(parent, property).GetString() ?? throw new InvalidDataException("shader manifest: null '" + property + "'"); + + private static int Int(JsonElement parent, string property) => Get(parent, property).GetInt32(); + + private static JsonElement.ArrayEnumerator Arr(JsonElement parent, string property) + { + JsonElement value = Get(parent, property); + if (value.ValueKind != JsonValueKind.Array) + { + throw new InvalidDataException("shader manifest: '" + property + "' is not an array"); + } + return value.EnumerateArray(); + } + + private static List Strings(JsonElement parent, string property) + { + var list = new List(); + foreach (JsonElement e in Arr(parent, property)) list.Add(e.GetString() ?? ""); + return list; + } +} + +internal sealed class NativeProgram +{ + /// The program's PassName, which is also its source file base name. + public string Name = ""; + /// The variant axes the source branches on, sorted. + public List Axes = new(); + /// One per combination of values, sorted by key. + public List Variants = new(); +} + +internal sealed class NativeVariant +{ + /// : sorted NAME=value, comma separated. + public string Key = ""; + public List Stages = new(); + /// The push-constant block; null when the program declares none. + public NativeBlock? Push; + /// The program record at set 2, OPTIMUM_BINDING_PROGRAM_RECORD; null when none. + public NativeBlock? Record; + /// FrameGlobals members the program reads under their own name (owner include rule), in block order. + public List FrameMembers = new(); + /// Bindless sampler slots, in push-block (GLSL 330 declaration) order. + public List Samplers = new(); + /// The fixed set 0 textures the shipped modules sample. + public List FrameTextures = new(); + public List StorageBindings = new(); + public List VertexInputs = new(); + public List FragmentOutputs = new(); + /// Bit n set when the shipped fragment module stores to the output at location n. + public uint WrittenOutputs; + public List SpecializationConstants = new(); +} + +internal sealed class NativeStage +{ + /// vertex or fragment. + public string Stage = ""; + /// Source file name relative to the source directory. + public string Source = ""; + /// SPIR-V file name relative to the manifest's directory. + public string Spirv = ""; + /// Lower-case hex SHA-256 of the SPIR-V file. + public string Sha256 = ""; +} + +internal sealed class NativeBlock +{ + public string TypeName = ""; + public int Size; + public List Members = new(); +} + +internal sealed class NativeMember +{ + public string Name = ""; + public string Type = ""; + public int Offset; + public int Size; + /// 0 when not an array, -1 for a runtime-sized array. + public int ArrayLength; +} + +internal sealed class NativeSampler +{ + /// The GLSL 330 sampler name, which is also the push member's name. + public string Name = ""; + public string GlslType = ""; + /// The set 1 array the slot indexes (optimumTextures2D, ...). + public string BindlessArray = ""; + public int ArrayBinding; + /// Offset of the uint slot index in the push block. + public int PushOffset; + /// Position among the program's sampler slots; matches the GLSL 330 declaration order by authoring. + public int Order; +} + +internal sealed class NativeFrameTexture +{ + public string Name = ""; + public string GlslType = ""; + public int Binding; +} + +internal sealed class NativeStorageBinding +{ + public string Name = ""; + public int Set; + public int Binding; + /// storageBuffer or uniformBuffer. + public string DescriptorType = ""; + public int ArrayLength; + public bool RuntimeArray; + /// Whether a shipped (optimised) module still reads it. + public bool Used; +} + +internal sealed class NativeInterfaceVariable +{ + public int Location; + public string Name = ""; + public string Type = ""; + public int ArrayLength; +} + +internal sealed class NativeSpecConstant +{ + public int Id; + public string Name = ""; + /// bool, int, uint, float or double. + public string Type = ""; + /// The declared default; a bool is 0 or 1. + public double Default; +} diff --git a/Optimum.Render.Vulkan/Shaders/ProgramInterfaceLayout.cs b/Optimum.Render.Vulkan/Shaders/ProgramInterfaceLayout.cs new file mode 100644 index 00000000..3086226d --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/ProgramInterfaceLayout.cs @@ -0,0 +1,979 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using Optimum.Render.Vulkan.Core; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// One vertex input a program declares, and where it lives. +internal readonly record struct VertexInputSlot(string Name, int Location, GlslType Type); + +/// One member of the program record (the generated default-uniform block). +internal sealed class UniformMember +{ + public string Name = ""; + public GlslType Type; + /// 0 when the member is not an array. + public int ArrayLength; + /// Byte offset into the block. + public int Offset; + /// Total bytes, counting every array element. + public int Size; + /// Default value as written in the shader, or null. + public string? Initializer; + + public int ElementCount => ArrayLength == 0 ? 1 : ArrayLength; +} + +/// +/// One sampler a program declares. Under the shared pipeline layout (plan decision 9) +/// a sampler is either one of set 0's fixed frame textures, read under its own name, +/// or a slot index into set 1's bindless array of its kind, carried in the push block +/// under the sampler's name. +/// +internal sealed class SamplerBinding +{ + public string Name = ""; + public string TypeName = ""; + + /// + /// Declaration order across the program, vertex stage first: the texture unit the + /// client's own bookkeeping (ShaderProgram.collectUniformNames) assigns by default. + /// + public int Order; + + /// The set 0 binding when this is a fixed frame texture (), else -1. + public int FrameBinding = -1; + + /// The set 1 array the slot indexes; meaningful only when is false. + public TextureKind Kind; + + /// Byte offset of the slot index in the push block, or -1 for a frame texture. + public int PushOffset = -1; + + public bool IsFrameTexture => FrameBinding >= 0; +} + +/// A named uniform or storage block and its set 2 binding. +internal sealed class BlockBinding +{ + public string BlockName = ""; + public int Set = SetConvention.StorageSet; + public int Binding; +} + +/// +/// The complete interface of a linked program: uniforms, samplers, blocks, and +/// the location assignments for every stage boundary. +/// +/// All of it has to be resolved per program rather than per stage, because GL +/// links by name and Vulkan links by number. Two consequences drive the design: +/// +/// A uniform named in two stages is one uniform in GL - zNear is declared +/// in both the vertex and fragment shader and carries one value - so the backend +/// generates a single uniform block, byte-identical in every stage, whose members +/// are the union of what the stages declare. +/// +/// A varying has no location in GL, but SPIR-V requires one on every user-defined +/// input and output, and the vertex output and fragment input must agree. The +/// vanilla shaders declare bare out vec2 texCoord;, so the backend assigns +/// those numbers itself and hands the same assignment to both stages. +/// +/// This is why a stage cannot be compiled to SPIR-V alone, and why +/// CompileShader only stages work that LinkProgram finishes. +/// +internal sealed partial class ProgramInterfaceLayout +{ + public const string BlockTypeName = "OptimumUniforms"; + public const string PushBlockTypeName = "OptimumDraw"; + + /// Bytes of one sampler slot index in the push block. + public const int SlotBytes = 4; + + /// + /// The shared frame members each stage declared (see ). + /// Emitted per stage for the same reason is. + /// + public Dictionary> FrameMembersByStage { get; } = new(); + + /// + /// The array length each shared frame member was declared with in this program: + /// a shader may read a prefix of the shared array (pointLights[DYNLIGHTS]). + /// 0 for a member that is not an array. + /// + public Dictionary FrameMemberDeclaredLengths { get; } = new(StringComparer.Ordinal); + + /// Whether the program reads anything from the shared frame block. + public bool UsesFrameBlock => FrameMemberDeclaredLengths.Count > 0; + + /// Members in declaration order, vertex stage first. + public List Members { get; } = new(); + public Dictionary MembersByName { get; } = new(StringComparer.Ordinal); + + /// + /// Which members each stage actually declared. + /// + /// The generated block is emitted per stage rather than whole, because a name + /// that is a uniform in one stage can be something else entirely in another: + /// bilateralblur.vsh declares uniform vec2 frameSize while its + /// fragment shader declares in vec2 frameSize. Emitting the union into + /// both stages would redefine the varying. Members carry explicit offsets, so + /// each stage sees a subset of one shared buffer layout. + /// + public Dictionary> MembersByStage { get; } = new(); + + /// Every sampler, frame textures included, in declaration order. + public List Samplers { get; } = new(); + public Dictionary SamplersByName { get; } = new(StringComparer.Ordinal); + + /// + /// Which samplers each stage declared: a stage gets push members and body rewrites + /// only for its own, for the reason exists. + /// + public Dictionary> SamplersByStage { get; } = new(); + + /// Bytes of the push block: one slot index per non-frame sampler; 0 when there are none. + public int PushConstantSize { get; private set; } + + /// Whether any sampler reads a set 0 frame texture. + public bool UsesFrameTextures { get; private set; } + + /// Whether a draw of the program needs set 2: a record, a named block or a storage block. + public bool UsesStorageSet => HasUniformBlock || UniformBlocks.Count > 0 || StorageBlocks.Count > 0; + + public List UniformBlocks { get; } = new(); + public List StorageBlocks { get; } = new(); + + /// Stage-to-stage varying locations, keyed by variable name. + public Dictionary VaryingLocations { get; } = new(StringComparer.Ordinal); + + /// Vertex attribute locations for inputs that declared none. + public Dictionary VertexInputLocations { get; } = new(StringComparer.Ordinal); + + /// + /// Every vertex input the program declares, whatever supplies it. + /// + /// A shader routinely reads attributes the mesh does not carry - the GUI + /// quad has only positions and UVs, while gui.vsh also declares a colour, a + /// render-flags int, a damage effect and a joint id. GL answers those reads + /// with the constant generic attribute, so the draw is well defined; Vulkan + /// has no equivalent and the values are undefined. Knowing the full set is + /// what lets the device supply the same constants. + /// + public List VertexInputs { get; } = new(); + + /// Fragment output locations for outputs that declared none. + public Dictionary FragmentOutputLocations { get; } = new(StringComparer.Ordinal); + + /// + /// Colour locations the fragment shader actually assigns somewhere in its + /// body. GL leaves an enabled attachment alone when the shader never writes + /// its output (undefined by the spec, preserved by every driver we ship on); + /// Vulkan writes undefined values into it. The device zeroes the colour + /// write mask of every attachment outside this set so both backends keep + /// the attachment's previous contents - the SSAO G-buffer under a + /// fullscreen compose pass, for one. + /// + public HashSet WrittenFragmentOutputs { get; } = new(); + + /// Size of the program record in bytes; 0 when it has no members. + public int BlockSize { get; private set; } + + public bool HasUniformBlock => BlockSize > 0; + + /// Diagnostics that made the layout unusable. + public List Errors { get; } = new(); + public bool HasErrors => Errors.Count > 0; + + /// + /// Builds the shadow buffer the CPU writes uniforms into, pre-filled with any + /// initialisers the shaders declared. GL applies those defaults at link time + /// and shaders rely on it: final.fsh never assigns extraGamma unless + /// colour grading is active and expects the declared 1.0. + /// + public byte[] CreateShadowBuffer() + { + var buffer = new byte[Math.Max(BlockSize, 0)]; + foreach (UniformMember member in Members) + { + if (member.Initializer != null) + { + WriteInitializer(buffer, member); + } + } + return buffer; + } + + internal static void WriteInitializer(byte[] buffer, UniformMember member) + { + // Only scalar literal defaults are honoured. Every initialiser in the + // shipped shaders is one; a constructor expression would need an + // evaluator to be worth supporting. + if (member.ArrayLength != 0 || member.Type.ComponentCount != 1) return; + + string text = member.Initializer!.Trim(); + switch (member.Type.Name) + { + case "float": + if (float.TryParse(text, NumberStyles.Float, CultureInfo.InvariantCulture, out float f)) + { + BitConverter.TryWriteBytes(buffer.AsSpan(member.Offset), f); + } + break; + case "int": + case "uint": + if (int.TryParse(text, NumberStyles.Integer, CultureInfo.InvariantCulture, out int i)) + { + BitConverter.TryWriteBytes(buffer.AsSpan(member.Offset), i); + } + break; + case "bool": + BitConverter.TryWriteBytes(buffer.AsSpan(member.Offset), text == "true" ? 1 : 0); + break; + } + } + + /// + /// Unions the stages into one layout. Stages arrive in a fixed order - vertex, + /// fragment, geometry - so the result is deterministic and the SPIR-V cache + /// key is stable across runs. + /// + /// + /// Locations from IShaderProgram's BindAttribLocation map, for mods + /// that name attributes through the API instead of a layout qualifier. + /// + /// + /// The program's include files; a uniform whose owning include is among them + /// reads the shared frame block (). + /// + public static ProgramInterfaceLayout Build( + IReadOnlyList<(EnumShaderType Stage, ParsedShader Parsed)> stages, + IReadOnlyDictionary? declaredAttributes = null, + IReadOnlySet? includes = null) + { + var layout = new ProgramInterfaceLayout(); + int offset = 0; + int nextNamedBinding = SetConvention.NamedBlockFirstBinding; + + foreach ((EnumShaderType stage, ParsedShader parsed) in stages) + { + foreach (GlslDeclaration declaration in parsed.Declarations) + { + switch (declaration.Kind) + { + case GlslDeclarationKind.DefaultUniform: + AddDefaultUniform(layout, declaration, stage, includes, ref offset); + break; + case GlslDeclarationKind.OpaqueUniform: + AddSampler(layout, declaration, stage); + break; + case GlslDeclarationKind.UniformBlock: + AddBlock(layout, layout.UniformBlocks, declaration, ref nextNamedBinding); + break; + case GlslDeclarationKind.StorageBlock: + AddBlock(layout, layout.StorageBlocks, declaration, ref nextNamedBinding); + break; + } + } + } + + AssignInterfaceLocations(layout, stages, declaredAttributes); + + layout.BlockSize = offset; + return layout; + } + + // ------------------------------------------------------------------ uniforms + + private static void AddDefaultUniform( + ProgramInterfaceLayout layout, GlslDeclaration declaration, EnumShaderType stage, + IReadOnlySet? includes, ref int offset) + { + if (!GlslType.TryParse(declaration.TypeName, out GlslType type)) + { + // A struct-typed uniform, or a type this backend does not model. The + // rewriter leaves the declaration alone, so the shader still compiles; + // it simply is not settable through the generated block. + return; + } + + // A value Use() writes into every program that includes its owner: it reads + // the shared frame block instead of taking room in this program's own. + if (declaration.UnresolvedArraySize == null && + FrameGlobals.TryPlace(declaration.Name, type, declaration.ArrayLength, includes, out _)) + { + if (!layout.FrameMembersByStage.TryGetValue(stage, out HashSet? frameMembers)) + { + frameMembers = new HashSet(StringComparer.Ordinal); + layout.FrameMembersByStage[stage] = frameMembers; + } + frameMembers.Add(declaration.Name); + + // Every stage is compiled with the same defines, so the lengths agree; + // the longest is kept should they not, since each is a prefix. + if (!layout.FrameMemberDeclaredLengths.TryGetValue(declaration.Name, out int known) || + declaration.ArrayLength > known) + { + layout.FrameMemberDeclaredLengths[declaration.Name] = declaration.ArrayLength; + } + return; + } + + if (!layout.MembersByStage.TryGetValue(stage, out HashSet? stageMembers)) + { + stageMembers = new HashSet(StringComparer.Ordinal); + layout.MembersByStage[stage] = stageMembers; + } + stageMembers.Add(declaration.Name); + + if (declaration.UnresolvedArraySize != null) + { + layout.Errors.Add( + $"uniform '{declaration.Name}' has array size '{declaration.UnresolvedArraySize}' " + + $"which did not resolve to a constant in the {stage} stage"); + return; + } + + if (layout.MembersByName.TryGetValue(declaration.Name, out UniformMember? existing)) + { + // Declared in more than one stage. GL merges them; so do we, but only + // when they agree - a mismatch is a bug GL would reject at link time. + if (!existing.Type.Equals(type) || existing.ArrayLength != declaration.ArrayLength) + { + layout.Errors.Add( + $"uniform '{declaration.Name}' is declared as '{existing.Type.Name}' " + + $"and '{type.Name}' in different stages"); + } + return; + } + + offset = Align(offset, type.Alignment); + + var member = new UniformMember + { + Name = declaration.Name, + Type = type, + ArrayLength = declaration.ArrayLength, + Offset = offset, + Initializer = declaration.Initializer, + }; + member.Size = type.Size * member.ElementCount; + offset += member.Size; + + layout.Members.Add(member); + layout.MembersByName[member.Name] = member; + } + + /// + /// Classifies a sampler under the shared layout. A name and type that match one of + /// set 0's fixed frame textures read that binding; every other sampler takes the + /// next push-block slot, in declaration order, and indexes the set 1 array of its + /// GLSL type. A type set 1 has no array for, a sampler array, or more slots than + /// the push block holds is a link error. + /// + private static void AddSampler(ProgramInterfaceLayout layout, GlslDeclaration declaration, EnumShaderType stage) + { + if (!layout.SamplersByStage.TryGetValue(stage, out HashSet? stageSamplers)) + { + stageSamplers = new HashSet(StringComparer.Ordinal); + layout.SamplersByStage[stage] = stageSamplers; + } + stageSamplers.Add(declaration.Name); + + if (layout.SamplersByName.TryGetValue(declaration.Name, out SamplerBinding? existing)) + { + if (!string.Equals(existing.TypeName, declaration.TypeName, StringComparison.Ordinal)) + { + layout.Errors.Add($"sampler '{declaration.Name}' is declared as '{existing.TypeName}' " + + $"and '{declaration.TypeName}' in different stages"); + } + return; + } + + var binding = new SamplerBinding + { + Name = declaration.Name, + TypeName = declaration.TypeName, + Order = layout.Samplers.Count, + }; + layout.Samplers.Add(binding); + layout.SamplersByName[binding.Name] = binding; + + if (declaration.ArrayLength != 0 || declaration.UnresolvedArraySize != null) + { + layout.Errors.Add($"sampler '{declaration.Name}' is an array, which the shared layout's push slots cannot index"); + return; + } + + foreach (SetConvention.Binding frame in SetConvention.FrameTextures) + { + if (string.Equals(frame.Name, declaration.Name, StringComparison.Ordinal) && + string.Equals(frame.GlslType, declaration.TypeName, StringComparison.Ordinal)) + { + binding.FrameBinding = frame.Value; + layout.UsesFrameTextures = true; + return; + } + } + + if (!BindlessKinds.TryFromGlslType(declaration.TypeName, out TextureKind kind)) + { + layout.Errors.Add($"sampler '{declaration.Name}' has type '{declaration.TypeName}', " + + "for which set 1 has no bindless array"); + return; + } + + binding.Kind = kind; + binding.PushOffset = layout.PushConstantSize; + layout.PushConstantSize += SlotBytes; + if (layout.PushConstantSize > SetConvention.PushConstantBytes) + { + layout.Errors.Add($"sampler '{declaration.Name}' needs push byte {layout.PushConstantSize}, " + + $"past the {SetConvention.PushConstantBytes} the shared layout holds"); + } + } + + /// + /// Gives a named block its set 2 binding. The game's Animation and + /// AnimationPrev blocks take the convention's animation bindings, the first + /// storage block takes FaceData's, and every other block takes the next binding of + /// the named-block range in declaration order. A binding the shader stated is not + /// kept: chunkopaque.vsh's binding = 3 is the record's binding under the + /// shared layout, and the mesh path binds FaceData by the convention's number. + /// + private static void AddBlock( + ProgramInterfaceLayout layout, List blocks, GlslDeclaration declaration, ref int nextNamedBinding) + { + foreach (BlockBinding existing in blocks) + { + if (existing.BlockName == declaration.Name) return; + } + + int binding; + bool storage = declaration.Kind == GlslDeclarationKind.StorageBlock; + if (!storage && declaration.Name == "Animation" && !HasBinding(layout, SetConvention.AnimationBinding)) + { + binding = SetConvention.AnimationBinding; + } + else if (!storage && declaration.Name == "AnimationPrev" && !HasBinding(layout, SetConvention.AnimationPrevBinding)) + { + binding = SetConvention.AnimationPrevBinding; + } + else if (storage && !HasBinding(layout, SetConvention.FaceDataBinding)) + { + binding = SetConvention.FaceDataBinding; + } + else if (nextNamedBinding <= SetConvention.NamedBlockLastBinding) + { + binding = nextNamedBinding++; + } + else + { + layout.Errors.Add($"block '{declaration.Name}' does not fit set 2: the shared layout holds " + + $"{SetConvention.NamedBlockLastBinding - SetConvention.NamedBlockFirstBinding + 1} named blocks"); + return; + } + + blocks.Add(new BlockBinding { BlockName = declaration.Name, Binding = binding }); + } + + private static bool HasBinding(ProgramInterfaceLayout layout, int binding) + { + foreach (BlockBinding block in layout.UniformBlocks) if (block.Binding == binding) return true; + foreach (BlockBinding block in layout.StorageBlocks) if (block.Binding == binding) return true; + return false; + } + + // ----------------------------------------------------------------- locations + + /// + /// Assigns the numbers SPIR-V demands and GLSL 330 leaves implicit: vertex + /// attribute locations, stage-to-stage varying locations, and fragment output + /// locations. Explicit qualifiers already in the source always win, and the + /// generated numbers fill the gaps around them. + /// + private static void AssignInterfaceLocations( + ProgramInterfaceLayout layout, + IReadOnlyList<(EnumShaderType Stage, ParsedShader Parsed)> stages, + IReadOnlyDictionary? declaredAttributes) + { + var usedVertexInputs = new HashSet(); + var usedVaryings = new HashSet(); + var usedFragmentOutputs = new HashSet(); + // Declared outputs that the body never assigns (a G-buffer output kept + // under an #if that compiled out its store) are still declared; only + // outputs with a store count as written. + var fragmentOutputDeclarations = new List(); + string fragmentSource = ""; + + // Pass one: record every location the shaders stated outright. + foreach ((EnumShaderType stage, ParsedShader parsed) in stages) + { + foreach (GlslDeclaration declaration in parsed.Declarations) + { + if (declaration.Location < 0) continue; + + if (stage == EnumShaderType.VertexShader && declaration.Kind == GlslDeclarationKind.Input) + { + Occupy(usedVertexInputs, declaration.Location, LocationSpan(declaration)); + RecordVertexInput(layout, declaration, declaration.Location); + } + else if (stage == EnumShaderType.FragmentShader && declaration.Kind == GlslDeclarationKind.Output) + { + Occupy(usedFragmentOutputs, declaration.Location, LocationSpan(declaration)); + fragmentOutputDeclarations.Add(declaration); + fragmentSource = parsed.Source; + } + else + { + layout.VaryingLocations[declaration.Name] = declaration.Location; + Occupy(usedVaryings, declaration.Location, LocationSpan(declaration)); + } + } + } + + // Attribute locations bound through the API rather than the shader. + if (declaredAttributes != null) + { + foreach (KeyValuePair attribute in declaredAttributes) + { + layout.VertexInputLocations[attribute.Key] = attribute.Value; + Occupy(usedVertexInputs, attribute.Value, 1); + } + } + + // Pass two: fill in the rest. + foreach ((EnumShaderType stage, ParsedShader parsed) in stages) + { + foreach (GlslDeclaration declaration in parsed.Declarations) + { + if (declaration.Location >= 0) continue; + int span = LocationSpan(declaration); + + if (stage == EnumShaderType.VertexShader && declaration.Kind == GlslDeclarationKind.Input) + { + // A name already present came from declaredAttributes - bound + // through the API rather than the shader - and still needs + // recording, because the location is known but the type only + // appears here. + if (layout.VertexInputLocations.TryGetValue(declaration.Name, out int bound)) + { + RecordVertexInput(layout, declaration, bound); + continue; + } + int assigned = Reserve(usedVertexInputs, span); + layout.VertexInputLocations[declaration.Name] = assigned; + RecordVertexInput(layout, declaration, assigned); + } + else if (stage == EnumShaderType.FragmentShader && declaration.Kind == GlslDeclarationKind.Output) + { + if (layout.FragmentOutputLocations.ContainsKey(declaration.Name)) continue; + layout.FragmentOutputLocations[declaration.Name] = Reserve(usedFragmentOutputs, span); + fragmentOutputDeclarations.Add(declaration); + fragmentSource = parsed.Source; + } + else if (declaration.Kind is GlslDeclarationKind.Input or GlslDeclarationKind.Output) + { + // A varying. The first stage to mention the name fixes the + // number; the matching stage reads it back out of the map, so + // vertex out and fragment in always agree. + if (layout.VaryingLocations.ContainsKey(declaration.Name)) continue; + layout.VaryingLocations[declaration.Name] = Reserve(usedVaryings, span); + } + } + } + + foreach (GlslDeclaration declaration in fragmentOutputDeclarations) + { + int location = declaration.Location >= 0 + ? declaration.Location + : layout.FragmentOutputLocations.TryGetValue(declaration.Name, out int assignedLocation) ? assignedLocation : -1; + if (location < 0) continue; + if (!TryGetWrittenFragmentOutputElements(fragmentSource, declaration.Name, out HashSet? writtenElements)) + { + continue; + } + int span = LocationSpan(declaration); + // An index on a non-array output selects a component (or a matrix + // column), not an attachment, so it still writes the whole span. + if (writtenElements == null || declaration.ArrayLength == 0) + { + for (int i = 0; i < span; i++) layout.WrittenFragmentOutputs.Add(location + i); + continue; + } + + // Only some elements of an output array are stored to. Marking the + // whole span written would leave colour writes on for attachments + // the shader never touches, and Vulkan then writes undefined data + // into them (GL would have preserved the attachment). + int perElement = span / Math.Max(declaration.ArrayLength == 0 ? 1 : declaration.ArrayLength, 1); + perElement = Math.Max(perElement, 1); + foreach (int element in writtenElements) + { + for (int i = 0; i < perElement; i++) + { + int slot = location + element * perElement + i; + if (slot < location + span) layout.WrittenFragmentOutputs.Add(slot); + } + } + } + } + + /// + /// Whether the fragment body stores to : a plain, + /// swizzled or indexed assignment, or a compound one. Declarations are + /// excluded by requiring the identifier not to be preceded by a type or + /// the "out" keyword on the same statement. + /// + internal static bool FragmentOutputIsAssigned(string source, string name) + => TryGetWrittenFragmentOutputElements(source, name, out _); + + /// + /// Which elements of a fragment output the body stores to. + /// Returns false when nothing stores to it at all. On true, + /// is null when the whole variable is written - + /// a plain or swizzled store, or an index the parser cannot fold to a + /// constant - and otherwise holds the constant element indices that are. + /// + internal static bool TryGetWrittenFragmentOutputElements( + string source, string name, out HashSet? elements) + { + elements = null; + var store = new System.Text.RegularExpressions.Regex( + @"(?(); + foreach (System.Text.RegularExpressions.Match match in store.Matches(source)) + { + int statementStart = source.LastIndexOfAny(new[] { ';', '{', '}' }, Math.Max(match.Index - 1, 0)) + 1; + string before = source.Substring(statementStart, match.Index - statementStart); + if (System.Text.RegularExpressions.Regex.IsMatch(before, @"\bout\b|\bin\b|\buniform\b")) continue; + + assigned = true; + string suffix = match.Groups[1].Value; + if (!suffix.StartsWith("[", StringComparison.Ordinal)) + { + // Whole variable or a swizzle of it: everything is written. + elements = null; + return true; + } + + int close = suffix.IndexOf(']'); + string index = close < 0 ? "" : suffix.Substring(1, close - 1).Trim(); + if (!int.TryParse(index, System.Globalization.NumberStyles.Integer, + System.Globalization.CultureInfo.InvariantCulture, out int element) + || element < 0) + { + // Dynamic index: assume every element can be written. + elements = null; + return true; + } + indices.Add(element); + } + + if (!assigned) return false; + elements = indices; + return true; + } + + + /// + /// How many consecutive locations a variable consumes. A vector of any width + /// fits in one; a matrix takes one per column; an array multiplies by its + /// length. + /// + /// + /// Notes a vertex input so the device can supply GL's constant default when + /// the mesh does not carry it. Arrays and matrices are skipped: nothing in + /// the game declares one as a vertex input, and spanning several locations + /// would need a default per column rather than per attribute. + /// + private static void RecordVertexInput( + ProgramInterfaceLayout layout, GlslDeclaration declaration, int location) + { + if (location < 0) return; + if (declaration.ArrayLength != 0) return; + if (!GlslType.TryParse(declaration.TypeName, out GlslType type)) return; + if (type.IsMatrix || type.IsOpaque) return; + + foreach (VertexInputSlot existing in layout.VertexInputs) + { + if (existing.Location == location) return; + } + layout.VertexInputs.Add(new VertexInputSlot(declaration.Name, location, type)); + } + + private static int LocationSpan(GlslDeclaration declaration) + { + int elements = declaration.ArrayLength == 0 ? 1 : declaration.ArrayLength; + int perElement = GlslType.TryParse(declaration.TypeName, out GlslType type) && type.IsMatrix + ? type.Columns + : 1; + return Math.Max(1, elements * perElement); + } + + private static void Occupy(HashSet used, int start, int span) + { + for (int i = 0; i < span; i++) used.Add(start + i); + } + + private static int Reserve(HashSet used, int span) + { + int candidate = 0; + while (true) + { + bool free = true; + for (int i = 0; i < span; i++) + { + if (used.Contains(candidate + i)) { free = false; break; } + } + if (free) + { + Occupy(used, candidate, span); + return candidate; + } + candidate++; + } + } + + private static int Align(int value, int alignment) => + alignment <= 1 ? value : (value + alignment - 1) / alignment * alignment; +} + +/// +/// The layout of a program linked from the native manifest (docs/vulkan.md). +/// +/// The draw path reads the same whichever way a program was +/// linked, so a native program is described in the rewriter's terms: frame members through the owner +/// rule (the manifest's frameMembers), record members at their reflected offsets, sampler slots +/// at their push offsets, and set 0 frame textures under their game names. The one thing the rewriter +/// never produces is a push member that is not a sampler slot (a DRAW uniform, section 4); those live in +/// and are written through the push location range. +/// +internal sealed partial class ProgramInterfaceLayout +{ + /// Push-block members that are not sampler slots, in block order. Empty for a rewritten program. + public List PushMembers { get; } = new(); + public Dictionary PushMembersByName { get; } = new(StringComparer.Ordinal); + + /// + /// The per-program push shadow a native program's non-slot push members persist in, seeded with the + /// GLSL 330 initializers; null when the push block holds only sampler slots (every rewritten program), + /// because the draw rewrites every slot anyway. + /// + public byte[]? CreatePushShadow() + { + if (PushMembers.Count == 0) return null; + var buffer = new byte[PushConstantSize]; + foreach (UniformMember member in PushMembers) + { + if (member.Initializer != null) WriteInitializer(buffer, member); + } + return buffer; + } + + /// + /// Builds the layout of one manifest variant. is what the GLSL 330 source the + /// program still carries says: sampler units in collectUniformNames order and the declarations' + /// initializers, neither of which the manifest records. + /// + public static ProgramInterfaceLayout FromNative(NativeVariant variant, GlslUniformOracle oracle) + { + var layout = new ProgramInterfaceLayout(); + + foreach (string name in variant.FrameMembers) + { + if (!FrameGlobals.TryGetMember(name, out UniformMember frame)) + { + layout.Errors.Add("frame member '" + name + "' is not in FrameGlobals"); + continue; + } + layout.FrameMemberDeclaredLengths[name] = frame.ArrayLength; + } + + if (variant.Record != null) + { + foreach (NativeMember member in variant.Record.Members) + { + UniformMember? built = MemberOf(layout, member, oracle, "record"); + if (built == null) continue; + layout.Members.Add(built); + layout.MembersByName[built.Name] = built; + } + layout.BlockSize = variant.Record.Size; + } + + var slots = new Dictionary(StringComparer.Ordinal); + foreach (NativeSampler sampler in variant.Samplers) slots[sampler.Name] = sampler; + + if (variant.Push != null) + { + if (variant.Push.Size > SetConvention.PushConstantBytes) + { + layout.Errors.Add("push block of " + variant.Push.Size + " B is past the " + SetConvention.PushConstantBytes + " B the shared layout holds"); + } + foreach (NativeMember member in variant.Push.Members) + { + if (slots.ContainsKey(member.Name)) continue; + UniformMember? built = MemberOf(layout, member, oracle, "push"); + if (built == null) continue; + layout.PushMembers.Add(built); + layout.PushMembersByName[built.Name] = built; + } + layout.PushConstantSize = variant.Push.Size; + } + + AddNativeSamplers(layout, variant, slots, oracle); + + foreach (NativeStorageBinding binding in variant.StorageBindings) + { + if (!binding.Used) continue; + if (binding.Set != SetConvention.StorageSet) + { + layout.Errors.Add("storage binding '" + binding.Name + "' is in set " + binding.Set); + continue; + } + // The client names its animation UBOs "Animation" and "AnimationPrev"; the device feeds a block + // binding from the UBO bound under the block's name, so the convention's binding decides the name. + switch (binding.Binding) + { + case SetConvention.FaceDataBinding: + layout.StorageBlocks.Add(new BlockBinding { BlockName = binding.Name, Binding = binding.Binding }); + break; + case SetConvention.AnimationBinding: + layout.UniformBlocks.Add(new BlockBinding { BlockName = "Animation", Binding = binding.Binding }); + break; + case SetConvention.AnimationPrevBinding: + layout.UniformBlocks.Add(new BlockBinding { BlockName = "AnimationPrev", Binding = binding.Binding }); + break; + default: + layout.Errors.Add("storage binding '" + binding.Name + "' at binding " + binding.Binding + + " has no native feed (the named-block range is the rewriter's)"); + break; + } + } + + foreach (NativeInterfaceVariable input in variant.VertexInputs) + { + if (input.ArrayLength != 0 || !GlslType.TryParse(input.Type, out GlslType type) || type.IsMatrix || type.IsOpaque) continue; + layout.VertexInputLocations[input.Name] = input.Location; + RecordNativeVertexInput(layout, new VertexInputSlot(input.Name, input.Location, type)); + } + + foreach (NativeInterfaceVariable output in variant.FragmentOutputs) + { + layout.FragmentOutputLocations[output.Name] = output.Location; + } + for (int bit = 0; bit < 32; bit++) + { + if ((variant.WrittenOutputs & (1u << bit)) != 0) layout.WrittenFragmentOutputs.Add(bit); + } + + return layout; + } + + private static void RecordNativeVertexInput(ProgramInterfaceLayout layout, VertexInputSlot slot) + { + foreach (VertexInputSlot existing in layout.VertexInputs) + { + if (existing.Location == slot.Location) return; + } + layout.VertexInputs.Add(slot); + } + + private static UniformMember? MemberOf(ProgramInterfaceLayout layout, NativeMember member, GlslUniformOracle oracle, string block) + { + if (!GlslType.TryParse(member.Type, out GlslType type)) + { + layout.Errors.Add(block + " member '" + member.Name + "' has type '" + member.Type + "', which the device does not model"); + return null; + } + if (member.ArrayLength < 0) + { + layout.Errors.Add(block + " member '" + member.Name + "' is a runtime array"); + return null; + } + return new UniformMember + { + Name = member.Name, + Type = type, + ArrayLength = member.ArrayLength, + Offset = member.Offset, + Size = member.Size, + Initializer = oracle.InitializerOf(member.Name), + }; + } + + /// + /// The sampler list in the texture-unit order the client assigns (collectUniformNames over the + /// GLSL 330 text): a name that is a set 0 frame texture reads that binding, a manifest slot takes its + /// push offset, and a name the oracle sees but neither side declares (its Array quirk) still + /// consumes its unit. A slot the oracle cannot see (a type outside its list) follows, in push order. + /// + private static void AddNativeSamplers( + ProgramInterfaceLayout layout, NativeVariant variant, Dictionary slots, GlslUniformOracle oracle) + { + var placed = new HashSet(StringComparer.Ordinal); + int nextUnit = 0; + foreach ((string name, string typeName, int unit) in oracle.SamplersByUnit()) + { + nextUnit = Math.Max(nextUnit, unit + 1); + if (!placed.Add(name)) continue; + + if (slots.TryGetValue(name, out NativeSampler? slot)) + { + AddSlot(layout, slot, unit); + continue; + } + foreach (SetConvention.Binding frame in SetConvention.FrameTextures) + { + if (string.Equals(frame.Name, name, StringComparison.Ordinal) && + string.Equals(frame.GlslType, typeName, StringComparison.Ordinal)) + { + AddFrameTexture(layout, name, typeName, frame.Value, unit); + break; + } + } + } + + foreach (NativeSampler slot in variant.Samplers) + { + if (placed.Add(slot.Name)) AddSlot(layout, slot, nextUnit++); + } + foreach (NativeFrameTexture texture in variant.FrameTextures) + { + if (placed.Add(texture.Name)) AddFrameTexture(layout, texture.Name, texture.GlslType, texture.Binding, nextUnit++); + } + + layout.Samplers.Sort((a, b) => a.Order.CompareTo(b.Order)); + } + + private static void AddSlot(ProgramInterfaceLayout layout, NativeSampler slot, int unit) + { + if (!BindlessKinds.TryFromGlslType(slot.GlslType, out TextureKind kind)) + { + layout.Errors.Add("sampler '" + slot.Name + "' has type '" + slot.GlslType + "', for which set 1 has no bindless array"); + return; + } + var binding = new SamplerBinding + { + Name = slot.Name, + TypeName = slot.GlslType, + Order = unit, + Kind = kind, + PushOffset = slot.PushOffset, + }; + layout.Samplers.Add(binding); + layout.SamplersByName[binding.Name] = binding; + } + + private static void AddFrameTexture(ProgramInterfaceLayout layout, string name, string typeName, int frameBinding, int unit) + { + var binding = new SamplerBinding { Name = name, TypeName = typeName, Order = unit, FrameBinding = frameBinding }; + layout.Samplers.Add(binding); + layout.SamplersByName[name] = binding; + layout.UsesFrameTextures = true; + } +} diff --git a/Optimum.Render.Vulkan/Shaders/SetConvention.cs b/Optimum.Render.Vulkan/Shaders/SetConvention.cs new file mode 100644 index 00000000..dbec6c13 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/SetConvention.cs @@ -0,0 +1,162 @@ + + +namespace Optimum.Render.Vulkan.Shaders; + +/// +/// The descriptor set convention of plan decision 9: one pipeline layout shared by +/// every program. Mirrors sources/shaders-vk/include/bindings.glsl, the +/// source of truth for native shaders. ShaderDeliveryTests checks compiled bindings +/// against the actual shared pipeline layout. +/// +/// | Set | Update | Contents | +/// | 0 frame | once per frame | FrameGlobals UBO (dynamic offset) and the fixed frame textures | +/// | 1 textures | when a texture is created or retired | bindless combined-image-sampler arrays, one per GLSL sampled type, PARTIALLY_BOUND and UPDATE_AFTER_BIND | +/// | 2 storage | per draw | FaceData, the animation buffers, the program record and named blocks | +/// | push | per draw | texture slot indices and per-draw scalars, at most | +/// +/// Array sizes are docs/vulkan.md#bindless-descriptors's starting sizes; the device +/// floor () is their sum. +/// +internal static class SetConvention +{ + public const string IncludePath = "sources/shaders-vk/include/bindings.glsl"; + + public const int FrameSet = 0; + public const int TextureSet = 1; + public const int StorageSet = 2; + public const int SetCount = 3; + + /// The spec minimum; everything a draw needs that is larger lives in a per-frame record addressed from here. + public const uint PushConstantBytes = 128; + + public const int FrameGlobalsBinding = 0; + + public const uint Texture2DCapacity = 16384; + public const uint Texture2DArrayCapacity = 1024; + public const uint TextureCubeCapacity = 256; + public const uint Texture3DCapacity = 256; + public const uint UnsignedTexture2DCapacity = 1024; + public const uint SignedTexture2DCapacity = 256; + public const uint Shadow2DCapacity = 128; + public const uint Shadow2DArrayCapacity = 64; + public const uint ShadowCubeCapacity = 64; + + public const uint TextureArrayCapacityTotal = + Texture2DCapacity + Texture2DArrayCapacity + TextureCubeCapacity + Texture3DCapacity + + UnsignedTexture2DCapacity + SignedTexture2DCapacity + Shadow2DCapacity + Shadow2DArrayCapacity + + ShadowCubeCapacity; + + /// A binding declared once in bindings.glsl: its define, value and, for samplers, the declaration. + public readonly record struct Binding(string Define, int Value, string GlslType, string Name, uint Capacity); + + /// + /// Set 0's fixed frame textures, under the names the game's shaders already use + /// (fogandlight.fsh declares the two shadow maps as sampler2DShadow). + /// Binding 0 is the FrameGlobals block. + /// + public static readonly Binding[] FrameTextures = + { + new("OPTIMUM_BINDING_SHADOW_MAP_FAR", 1, "sampler2DShadow", "shadowMapFar", 1), + new("OPTIMUM_BINDING_SHADOW_MAP_NEAR", 2, "sampler2DShadow", "shadowMapNear", 1), + new("OPTIMUM_BINDING_SKY", 3, "sampler2D", "sky", 1), + new("OPTIMUM_BINDING_GLOW", 4, "sampler2D", "glow", 1), + new("OPTIMUM_BINDING_LIQUID_DEPTH", 5, "sampler2D", "liquidDepth", 1), + }; + + /// Set 1: one runtime-sized array per GLSL sampled type. + public static readonly Binding[] TextureArrays = + { + new("OPTIMUM_BINDING_TEXTURES_2D", 0, "sampler2D", "optimumTextures2D", Texture2DCapacity), + new("OPTIMUM_BINDING_TEXTURES_2D_ARRAY", 1, "sampler2DArray", "optimumTextures2DArray", Texture2DArrayCapacity), + new("OPTIMUM_BINDING_TEXTURES_CUBE", 2, "samplerCube", "optimumTexturesCube", TextureCubeCapacity), + new("OPTIMUM_BINDING_TEXTURES_3D", 3, "sampler3D", "optimumTextures3D", Texture3DCapacity), + new("OPTIMUM_BINDING_TEXTURES_2D_UINT", 4, "usampler2D", "optimumTextures2DUint", UnsignedTexture2DCapacity), + new("OPTIMUM_BINDING_TEXTURES_2D_INT", 5, "isampler2D", "optimumTextures2DInt", SignedTexture2DCapacity), + new("OPTIMUM_BINDING_TEXTURES_2D_SHADOW", 6, "sampler2DShadow", "optimumTextures2DShadow", Shadow2DCapacity), + new("OPTIMUM_BINDING_TEXTURES_2D_ARRAY_SHADOW", 7, "sampler2DArrayShadow", "optimumTextures2DArrayShadow", Shadow2DArrayCapacity), + new("OPTIMUM_BINDING_TEXTURES_CUBE_SHADOW", 8, "samplerCubeShadow", "optimumTexturesCubeShadow", ShadowCubeCapacity), + }; + + /// + /// Set 2's program record: every non-frame uniform that is not in the push block + /// (docs/vulkan.md), a dynamic uniform buffer whose offset + /// moves when the record changed. Kept out of because it + /// is a uniform buffer, not a storage buffer. + /// + public const int ProgramRecordBinding = 3; + + /// + /// Set 2's storage buffers. FaceData is the chunk shaders' faceDataBuf; the + /// animation pair holds the game's Animation and AnimationPrev blocks, + /// read as std140 storage buffers (named after the blocks the rewriter maps there). + /// + public static readonly Binding[] StorageBuffers = + { + new("OPTIMUM_BINDING_FACE_DATA", 0, "buffer", "faceDataBuf", 1), + new("OPTIMUM_BINDING_ANIMATION", 1, "buffer", "Animation", 1), + new("OPTIMUM_BINDING_ANIMATION_PREV", 2, "buffer", "AnimationPrev", 1), + }; + + public const int FaceDataBinding = 0; + public const int AnimationBinding = 1; + public const int AnimationPrevBinding = 2; + + /// + /// Set 2 bindings for every other named block a rewritten program declares, in + /// declaration order: a GLSL 330 uniform Block { ... } becomes a + /// layout(std140) readonly buffer here, so the client's std140 bytes are read + /// unchanged. A program with more named blocks than this range fails to link. + /// + public const int NamedBlockFirstBinding = 4; + public const int NamedBlockLastBinding = 7; + + /// Every set 2 binding: storage buffers, the record, and the named-block range. + public const int StorageSetBindingCount = NamedBlockLastBinding + 1; +} + +/// +/// The specialization constants of the native shaders (docs/vulkan.md). +/// Mirrors sources/shaders-vk/include/specialization.glsl, the source of truth for native +/// shaders. ShaderDeliveryTests checks the compiled constant ids, types and defaults. +/// +/// Every constant replaces one quality or code-path define: a native source branches on +/// if (OPTIMUM_BLOOM != 0) where the GLSL 330 source has #if BLOOM > 0, the +/// declarations the branch uses are unconditional, and a settings change becomes a pipeline-key +/// change instead of a recompile. The defines that change a declaration stay variant axes +/// (TAAMOTION, USEOIT, USESSBO, GREEDYMESH, ...) and are not here. +/// +/// Defaults are 0, the value an undefined macro has in a GLSL #if and the value the +/// game's own includes fall back to (fogandlight.vsh: #ifndef DYNLIGHTS / #define +/// DYNLIGHTS 0, the same for MINBRIGHT). The runtime always specializes every +/// constant from the program's prefix, so a default only decides an unspecialized pipeline. +/// +internal static class SpecializationConvention +{ + public const string IncludePath = "sources/shaders-vk/include/specialization.glsl"; + + /// One constant: its id, GLSL name and type, default, and the define it replaces. + public readonly record struct Constant(uint Id, string Name, string GlslType, string Default, string Define); + + public static readonly Constant[] Constants = + { + new(0, "OPTIMUM_FXAA", "int", "0", "FXAA"), + new(1, "OPTIMUM_SSAOLEVEL", "int", "0", "SSAOLEVEL"), + new(2, "OPTIMUM_NORMALVIEW", "int", "0", "NORMALVIEW"), + new(3, "OPTIMUM_BLOOM", "int", "0", "BLOOM"), + new(4, "OPTIMUM_GODRAYS", "int", "0", "GODRAYS"), + new(5, "OPTIMUM_FOAMEFFECT", "int", "0", "FOAMEFFECT"), + new(6, "OPTIMUM_SHINYEFFECT", "int", "0", "SHINYEFFECT"), + new(7, "OPTIMUM_SHADOWQUALITY", "int", "0", "SHADOWQUALITY"), + new(8, "OPTIMUM_WAVINGSTUFF", "int", "0", "WAVINGSTUFF"), + new(9, "OPTIMUM_MINBRIGHT", "float", "0.0", "MINBRIGHT"), + new(10, "OPTIMUM_GREEDYMESH_GRAD", "int", "0", "GREEDYMESH_GRAD"), + // DYNLIGHTS no longer sizes an array (the frame block's point-light arrays are fixed at + // FrameGlobals.MaxDynamicLights and pointLightQuantity bounds the loop), but its zero + // value still selects fogandlight.vsh's no-point-light path, which also skips the night + // vision, MINBRIGHT and contrast terms. Keeping that path needs the value. + new(11, "OPTIMUM_DYNLIGHTS", "int", "0", "DYNLIGHTS"), + // Optimum AO (docs/vulkan.md#ambient-occlusion C.5, C.11): gates the class-channel writes and + // scene-ssao's GTAO compose branch, never an output or a varying. + new(12, "OPTIMUM_OPTIMUMAO", "int", "0", "OPTIMUMAO"), + }; +} diff --git a/Optimum.Render.Vulkan/Shaders/ShaderBinaryCache.cs b/Optimum.Render.Vulkan/Shaders/ShaderBinaryCache.cs new file mode 100644 index 00000000..95a1dc48 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/ShaderBinaryCache.cs @@ -0,0 +1,126 @@ +using System; +using System.IO; +using System.Security.Cryptography; +using System.Text; +using System.Threading; +using Optimum.Render.Vulkan.Core; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// +/// Compiled SPIR-V on disk, addressed by everything that decides it. +/// +/// Without it every program is compiled from GLSL at every launch - the whole +/// vanilla set, every settings variant the session reaches, every mod shader. The +/// rewritten source of a stage, which already has its defines resolved and its +/// includes expanded, fully determines its SPIR-V for a given compiler build and set +/// of options. The key is a hash of exactly that: this format's version, the +/// compiler identity (), the stage and the +/// source. A mod that edits a shader changes the source and so misses, rather than +/// loading stale SPIR-V under a file name. Nothing about the program's layout is +/// stored: the layout is rebuilt from the source before the cache is consulted, and +/// the cache only replaces the one step that is expensive. +/// +/// Each file carries a small header - magic, format version, payload length and a +/// SHA-256 of the payload - checked before anything is handed to the driver. +/// Anything that fails the check is a miss, never an error: a truncated file from an +/// interrupted write, a zero-filled block, a file another tool dropped in the +/// directory. The miss recompiles and overwrites it. Design and sources: +/// docs/vulkan.md#caches §5 and "Design for this renderer" item 1. +/// +internal sealed class ShaderBinaryCache +{ + private const uint SpirvMagic = 0x07230203; + + /// "OSPV", little-endian. + internal const uint FileMagic = 0x5650534F; + + /// + /// Bumped whenever the file layout changes, or whenever something that decides the + /// SPIR-V for a given source changes without appearing in the key. + /// + internal const uint FormatVersion = 1; + + /// Magic, format version, payload length, SHA-256 of the payload. + internal const int HeaderSize = 4 + 4 + 4 + 32; + + private long _hits; + private long _misses; + private long _writeFailures; + + public string Directory { get; } + + public long Hits => Interlocked.Read(ref _hits); + public long Misses => Interlocked.Read(ref _misses); + public long WriteFailures => Interlocked.Read(ref _writeFailures); + + public ShaderBinaryCache(string directory) + { + Directory = directory; + } + + /// The key for one stage's source under one compiler identity: lowercase hex SHA-256. + public static string KeyFor(string code, EnumShaderType stage, string compilerIdentity) + { + using var hash = IncrementalHash.CreateHash(HashAlgorithmName.SHA256); + hash.AppendData(Encoding.UTF8.GetBytes( + "optimum-spirv-" + FormatVersion + "\n" + compilerIdentity + "\n" + (int)stage + "\n")); + hash.AppendData(Encoding.UTF8.GetBytes(code)); + return Convert.ToHexStringLower(hash.GetHashAndReset()); + } + + /// The cached module for , or null. + public byte[]? TryGet(string key) + { + byte[]? file = CacheFileWriter.TryReadAll(PathFor(key)); + byte[]? spirv = file == null ? null : Unwrap(file); + Interlocked.Increment(ref spirv == null ? ref _misses : ref _hits); + return spirv; + } + + /// Stores a module. A failure to write only costs the next launch a compile. + public void Put(string key, byte[] spirv) + { + if (!IsSpirv(spirv)) return; + if (!CacheFileWriter.WriteAtomically(PathFor(key), Wrap(spirv))) + { + Interlocked.Increment(ref _writeFailures); + } + } + + internal static byte[] Wrap(byte[] spirv) + { + var file = new byte[HeaderSize + spirv.Length]; + BitConverter.TryWriteBytes(file.AsSpan(0), FileMagic); + BitConverter.TryWriteBytes(file.AsSpan(4), FormatVersion); + BitConverter.TryWriteBytes(file.AsSpan(8), (uint)spirv.Length); + SHA256.HashData(spirv, file.AsSpan(12, 32)); + spirv.CopyTo(file, HeaderSize); + return file; + } + + /// The payload of a well-formed file, or null for anything else. + internal static byte[]? Unwrap(byte[] file) + { + if (file.Length < HeaderSize) return null; + if (BitConverter.ToUInt32(file, 0) != FileMagic) return null; + if (BitConverter.ToUInt32(file, 4) != FormatVersion) return null; + if (BitConverter.ToUInt32(file, 8) != (uint)(file.Length - HeaderSize)) return null; + + ReadOnlySpan payload = file.AsSpan(HeaderSize); + Span hash = stackalloc byte[32]; + SHA256.HashData(payload, hash); + if (!hash.SequenceEqual(file.AsSpan(12, 32))) return null; + + byte[] spirv = payload.ToArray(); + return IsSpirv(spirv) ? spirv : null; + } + + /// A module header and a whole number of words; anything else is not SPIR-V. + internal static bool IsSpirv(byte[] data) => + data.Length >= 20 && data.Length % 4 == 0 && BitConverter.ToUInt32(data, 0) == SpirvMagic; + + // Two hex digits of fan-out keep a directory of a few thousand modules browsable. + private string PathFor(string key) => Path.Combine(Directory, key[..2], key + ".spv"); +} diff --git a/Optimum.Render.Vulkan/Shaders/ShaderCompiler.cs b/Optimum.Render.Vulkan/Shaders/ShaderCompiler.cs new file mode 100644 index 00000000..92c60384 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/ShaderCompiler.cs @@ -0,0 +1,497 @@ +using System; +using System.Globalization; +using System.Runtime.InteropServices; +using System.Text; +using Silk.NET.Shaderc; +using Vintagestory.API.Client; +using System.IO; +using System.Runtime.CompilerServices; + +namespace Optimum.Render.Vulkan.Shaders; + +/// Outcome of a preprocess or compile step. +internal sealed class ShaderCompileResult +{ + public bool Success; + public string? Error; + public string PreprocessedText = ""; + public byte[] Spirv = Array.Empty(); +} + +/// +/// Wraps shaderc for the two jobs the backend needs: resolving the preprocessor +/// before the rewriter looks at the source, and turning rewritten GLSL into +/// SPIR-V. +/// +/// Preprocessing is a separate pass on purpose. The game builds a block of +/// #defines per program - FXAA, SSAOLEVEL, SHADOWQUALITY, DYNLIGHTS and a +/// dozen more - and the shaders wrap declarations in #if on them, so the +/// set of uniforms a stage declares is not knowable until the conditionals are +/// resolved. Running a real preprocessor first means the rewriter only ever sees +/// straight-line declarations and never has to reason about conditionals. +/// +internal sealed unsafe partial class ShaderCompiler : IDisposable +{ + private readonly Shaderc _api; + private readonly Compiler* _compiler; + private bool _disposed; + + public ShaderCompiler() + { + _api = Shaderc.GetApi(); + _compiler = _api.CompilerInitialize(); + if (_compiler == null) + { + throw new InvalidOperationException("shaderc failed to initialise"); + } + } + + /// + /// Splices the program's #define prefix in after the version line and + /// resolves the preprocessor, exactly where and how the OpenGL path does it + /// in ClientPlatformWindows.CompileShader. + /// + public ShaderCompileResult Preprocess(string code, string prefixCode, string filename, EnumShaderType stage) + { + string spliced = SplicePrefix(RaiseVersionForPreprocessing(code), prefixCode); + var result = new ShaderCompileResult(); + + CompileOptions* options = CreateOptions(); + try + { + CompilationResult* compiled = CompileWith( + spliced, filename, stage, options, preprocessOnly: true); + try + { + if (!Succeeded(compiled, out string? error)) + { + result.Error = error; + return result; + } + + result.PreprocessedText = ReadBytesAsText(compiled); + result.Success = true; + return result; + } + finally + { + _api.ResultRelease(compiled); + } + } + finally + { + _api.CompileOptionsRelease(options); + } + } + + /// + /// Where compiled modules are kept between launches; null compiles every time. + /// The filename only names errors (no debug info is emitted), so it is not part of the key. + /// + public ShaderBinaryCache? BinaryCache { get; set; } + + /// + /// The options sets, spelled out for the cache key. + /// Change both together. + /// + internal const string OptionsIdentity = "glsl;vulkan1.3;spirv1.5;performance;no-debug-info;no-include-resolver"; + + private static string? _nativeIdentity; + + /// + /// Everything besides the source that decides the SPIR-V: the options and the + /// shaderc build. The C API exposes no compiler version, so the build is the + /// loaded library's own hash (docs/vulkan.md#caches §5). + /// + public string Identity => OptionsIdentity + ";" + (_nativeIdentity ??= NativeLibraryIdentity()); + + internal static string NativeLibraryIdentity() + { + string name = OperatingSystem.IsWindows() ? "shaderc_shared.dll" + : OperatingSystem.IsMacOS() ? "libshaderc_shared.dylib" + : "libshaderc_shared.so"; + string runtime = RuntimeInformation.RuntimeIdentifier; + + foreach (string? directory in new[] + { + AppContext.BaseDirectory, + System.IO.Path.GetDirectoryName(typeof(ShaderCompiler).Assembly.Location), + }) + { + if (string.IsNullOrEmpty(directory)) continue; + foreach (string candidate in new[] + { + System.IO.Path.Combine(directory, name), + System.IO.Path.Combine(directory, "runtimes", runtime, "native", name), + }) + { + try + { + if (!System.IO.File.Exists(candidate)) continue; + using System.IO.FileStream stream = System.IO.File.OpenRead(candidate); + return "shaderc-sha256:" + + Convert.ToHexStringLower(System.Security.Cryptography.SHA256.HashData(stream)); + } + catch (Exception error) when (error is System.IO.IOException or UnauthorizedAccessException) + { + } + } + } + + // Not found where the packagers put it: fall back to the binding's version, + // which pins the native package it ships with. + return "silk-shaderc-" + typeof(Shaderc).Assembly.GetName().Version; + } + + /// Compiles already-rewritten Vulkan GLSL to SPIR-V. + public ShaderCompileResult Compile(string code, string filename, EnumShaderType stage) + { + string? key = null; + if (BinaryCache != null) + { + key = ShaderBinaryCache.KeyFor(code, stage, Identity); + byte[]? cached = BinaryCache.TryGet(key); + if (cached != null) return new ShaderCompileResult { Success = true, Spirv = cached }; + } + + ShaderCompileResult compiled = CompileUncached(code, filename, stage); + if (key != null && compiled.Success) BinaryCache!.Put(key, compiled.Spirv); + return compiled; + } + + /// + /// Compiles with optimisation off and never through the cache. The optimiser strips every + /// OpName and drops declarations nothing uses, so the offline shader compiler reflects + /// names and declared interfaces from this twin of the shipped module + /// (docs/vulkan.md). Never shipped. + /// + public ShaderCompileResult CompileForReflection(string code, string filename, EnumShaderType stage) => + CompileUncached(code, filename, stage, optimize: false); + + /// + /// Stage tag for compute modules in the binary cache key: GL_COMPUTE_SHADER, which no + /// client stage uses, so a compute module never shares a key with a vertex or fragment one. + /// + internal const EnumShaderType ComputeStageTag = (EnumShaderType)37305; + + /// + /// Compiles a native Vulkan GLSL compute shader (sources/shaders-vk/**.comp) + /// to SPIR-V. No prefix, no rewriter: native shaders are written for the backend. + /// + public ShaderCompileResult CompileCompute(string code, string filename) + { + string? key = null; + if (BinaryCache != null) + { + key = ShaderBinaryCache.KeyFor(code, ComputeStageTag, Identity); + byte[]? cached = BinaryCache.TryGet(key); + if (cached != null) return new ShaderCompileResult { Success = true, Spirv = cached }; + } + + ShaderCompileResult compiled = CompileUncached(code, filename, ComputeStageTag); + if (key != null && compiled.Success) BinaryCache!.Put(key, compiled.Spirv); + return compiled; + } + + private ShaderCompileResult CompileUncached(string code, string filename, EnumShaderType stage, bool optimize = true) + { + CompileOptions* options = CreateOptions(); + if (!optimize) _api.CompileOptionsSetOptimizationLevel(options, OptimizationLevel.Zero); + try + { + CompilationResult* compiled = CompileWith(code, filename, stage, options, preprocessOnly: false); + try + { + return ReadCompiledSpirv(compiled); + } + finally + { + _api.ResultRelease(compiled); + } + } + finally + { + _api.CompileOptionsRelease(options); + } + } + + /// + /// Reproduces the vanilla splice: the prefix goes immediately after the + /// newline that ends the #version line, because a version directive + /// must be the first thing in a translation unit. + /// + internal static string SplicePrefix(string code, string prefixCode) + { + if (string.IsNullOrEmpty(prefixCode)) return code; + + int versionIndex = code.IndexOf("#version", StringComparison.Ordinal); + if (versionIndex < 0) return prefixCode + code; + + // A #version with nothing after it is a complete first line; the + // prefix follows it rather than displacing it. + int lineEnd = code.IndexOf('\n', versionIndex); + if (lineEnd < 0) return code + "\n" + prefixCode; + + return code.Insert(lineEnd + 1, prefixCode); + } + + /// + /// The lowest #version shaderc will preprocess for a SPIR-V target. + /// + /// The check runs during preprocessing, so a source below the floor is + /// rejected before the rewriter ever gets to raise it. Every vanilla shader + /// is well above this; it bites on the client's hardcoded 130 minimal-GUI + /// program and would bite on any mod shader written to an old version. + /// + private const int MinimumPreprocessVersion = 140; + + /// + /// Raises a below-floor #version to the version the rewriter targets + /// anyway, so preprocessing sees something shaderc will accept. + /// + /// Only sources below the floor are touched, which means no vanilla shader + /// changes at all. For the ones that do change, __VERSION__ becomes + /// 450 during preprocessing - a real difference, but the alternative is a + /// shader that cannot be compiled for this backend. + /// + internal static string RaiseVersionForPreprocessing(string code) + { + if (string.IsNullOrEmpty(code)) return code; + + int versionIndex = code.IndexOf("#version", StringComparison.Ordinal); + if (versionIndex < 0) return code; + + int numberStart = versionIndex + "#version".Length; + while (numberStart < code.Length && (code[numberStart] == ' ' || code[numberStart] == '\t')) + { + numberStart++; + } + int numberEnd = numberStart; + while (numberEnd < code.Length && char.IsAsciiDigit(code[numberEnd])) + { + numberEnd++; + } + if (numberEnd == numberStart) return code; + + if (!int.TryParse(code.AsSpan(numberStart, numberEnd - numberStart), + NumberStyles.Integer, CultureInfo.InvariantCulture, out int version)) + { + return code; + } + if (version >= MinimumPreprocessVersion) return code; + + return string.Concat( + code.AsSpan(0, numberStart), "450", code.AsSpan(numberEnd)); + } + + private CompileOptions* CreateOptions() + { + CompileOptions* options = _api.CompileOptionsInitialize(); + _api.CompileOptionsSetSourceLanguage(options, SourceLanguage.Glsl); + _api.CompileOptionsSetTargetEnv(options, TargetEnv.Vulkan, (uint)EnvVersion.Vulkan13); + _api.CompileOptionsSetTargetSpirv(options, SpirvVersion.Shaderc15); + _api.CompileOptionsSetOptimizationLevel(options, OptimizationLevel.Performance); + // The game resolves its own #include directives through ShaderRegistry + // before a stage ever reaches this class, so no include resolver is + // installed; an #include reaching shaderc is a genuine error. + return options; + } + + private CompilationResult* CompileWith( + string code, string filename, EnumShaderType stage, CompileOptions* options, bool preprocessOnly) + { + byte[] source = Encoding.UTF8.GetBytes(code); + byte[] name = Encoding.UTF8.GetBytes(filename ?? "shader"); + byte[] entry = Encoding.UTF8.GetBytes("main"); + + fixed (byte* sourcePtr = source) + fixed (byte* namePtr = name) + fixed (byte* entryPtr = entry) + { + ShaderKind kind = ToShaderKind(stage); + return preprocessOnly + ? _api.CompileIntoPreprocessedText( + _compiler, sourcePtr, (nuint)source.Length, kind, namePtr, entryPtr, options) + : _api.CompileIntoSpv( + _compiler, sourcePtr, (nuint)source.Length, kind, namePtr, entryPtr, options); + } + } + + private bool Succeeded(CompilationResult* result, out string? error) + { + if (result == null) + { + error = "shaderc returned no result"; + return false; + } + + if (_api.ResultGetCompilationStatus(result) == CompilationStatus.Success) + { + error = null; + return true; + } + + byte* message = _api.ResultGetErrorMessage(result); + error = message == null ? "unknown shaderc error" : Marshal.PtrToStringUTF8((IntPtr)message); + return false; + } + + private ShaderCompileResult ReadCompiledSpirv(CompilationResult* compiled) + { + var result = new ShaderCompileResult(); + if (!Succeeded(compiled, out string? error)) + { + result.Error = error; + return result; + } + + nuint length = _api.ResultGetLength(compiled); + byte* bytes = (byte*)_api.ResultGetBytes(compiled); + var spirv = new byte[(int)length]; + fixed (byte* destination = spirv) + { + Buffer.MemoryCopy(bytes, destination, spirv.Length, (long)length); + } + + result.Spirv = spirv; + result.Success = true; + return result; + } + + private string ReadBytesAsText(CompilationResult* result) + { + nuint length = _api.ResultGetLength(result); + byte* bytes = (byte*)_api.ResultGetBytes(result); + return length == 0 || bytes == null + ? "" + : Encoding.UTF8.GetString(bytes, (int)length); + } + + private static ShaderKind ToShaderKind(EnumShaderType stage) => stage switch + { + EnumShaderType.VertexShader => ShaderKind.VertexShader, + EnumShaderType.FragmentShader => ShaderKind.FragmentShader, + EnumShaderType.GeometryShader => ShaderKind.GeometryShader, + ComputeStageTag => ShaderKind.ComputeShader, + _ => ShaderKind.VertexShader, + }; + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + if (_compiler != null) + { + _api.CompilerRelease(_compiler); + } + _api.Dispose(); + } +} + +/// +/// Native shader sources (sources/shaders-vk, docs/vulkan.md) +/// resolve their own #include "file.glsl" through shaderc, unlike the game's GLSL 330 +/// programs, whose includes ShaderRegistry expands before the rewriter sees them. The +/// options are the same as 's plus an include resolver, and the result is +/// never cached: the cache key covers only the top-level text, not the files it pulls in. +/// +internal sealed unsafe partial class ShaderCompiler +{ + /// + /// Compiles a native GLSL 450 stage whose quoted includes resolve against the including + /// file's directory first and then . + /// + public ShaderCompileResult CompileNative(string code, string filename, EnumShaderType stage, string includeDirectory) + { + var resolver = new IncludeResolver(includeDirectory); + GCHandle handle = GCHandle.Alloc(resolver); + + CompileOptions* options = CreateOptions(); + try + { + _api.CompileOptionsSetIncludeCallbacks( + options, + new PfnIncludeResolveFn(&ResolveInclude), + new PfnIncludeResultReleaseFn(&ReleaseInclude), + (void*)GCHandle.ToIntPtr(handle)); + + CompilationResult* compiled = CompileWith(code, filename, stage, options, preprocessOnly: false); + try + { + return ReadCompiledSpirv(compiled); + } + finally + { + _api.ResultRelease(compiled); + } + } + finally + { + _api.CompileOptionsRelease(options); + handle.Free(); + } + } + + private sealed class IncludeResolver + { + private readonly string _directory; + + public IncludeResolver(string directory) => _directory = directory; + + /// The resolved path and text, or null and the reason. + public (string? Path, string Text) Resolve(string requested, string requesting, bool relative) + { + if (relative) + { + string? requestingDirectory = Path.GetDirectoryName(requesting); + if (!string.IsNullOrEmpty(requestingDirectory)) + { + string besideRequester = Path.GetFullPath(Path.Combine(requestingDirectory, requested)); + if (File.Exists(besideRequester)) return (besideRequester, File.ReadAllText(besideRequester)); + } + } + + string inDirectory = Path.GetFullPath(Path.Combine(_directory, requested)); + if (File.Exists(inDirectory)) return (inDirectory, File.ReadAllText(inDirectory)); + + return (null, "include '" + requested + "' not found in " + _directory); + } + } + + [UnmanagedCallersOnly(CallConvs = new[] { typeof(CallConvCdecl) })] + private static IncludeResult* ResolveInclude( + void* userData, byte* requestedSource, int type, byte* requestingSource, nuint includeDepth) + { + var resolver = (IncludeResolver)GCHandle.FromIntPtr((IntPtr)userData).Target!; + string requested = Marshal.PtrToStringUTF8((IntPtr)requestedSource) ?? ""; + string requesting = Marshal.PtrToStringUTF8((IntPtr)requestingSource) ?? ""; + + (string? path, string text) = resolver.Resolve(requested, requesting, relative: type == (int)IncludeType.Relative); + + // shaderc's convention: an empty source name marks a failure, and the content is the message. + var include = (IncludeResult*)NativeMemory.AllocZeroed((nuint)sizeof(IncludeResult)); + include->SourceName = CopyUtf8(path ?? "", out nuint nameLength); + include->SourceNameLength = nameLength; + include->Content = CopyUtf8(text, out nuint contentLength); + include->ContentLength = contentLength; + return include; + } + + [UnmanagedCallersOnly(CallConvs = new[] { typeof(CallConvCdecl) })] + private static void ReleaseInclude(void* userData, IncludeResult* include) + { + if (include == null) return; + NativeMemory.Free(include->SourceName); + NativeMemory.Free(include->Content); + NativeMemory.Free(include); + } + + private static byte* CopyUtf8(string text, out nuint length) + { + byte[] bytes = Encoding.UTF8.GetBytes(text); + var copy = (byte*)NativeMemory.Alloc((nuint)Math.Max(1, bytes.Length)); + bytes.AsSpan().CopyTo(new Span(copy, bytes.Length)); + length = (nuint)bytes.Length; + return copy; + } +} diff --git a/Optimum.Render.Vulkan/Shaders/ShaderRewriter.cs b/Optimum.Render.Vulkan/Shaders/ShaderRewriter.cs new file mode 100644 index 00000000..08d56ca5 --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/ShaderRewriter.cs @@ -0,0 +1,656 @@ +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Text; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// The result of rewriting one stage for Vulkan. +internal sealed class RewrittenShader +{ + public string Code = ""; + public List Errors { get; } = new(); + public bool HasErrors => Errors.Count > 0; +} + +/// +/// Turns a stage of GLSL 330 into GLSL 450 that glslang will accept for Vulkan. +/// +/// It works by editing spans of the original source rather than regenerating it. +/// Anything the parser did not classify is copied through byte for byte, so a +/// construct this backend has never seen - in a mod shader, say - survives intact +/// instead of being mangled. The failure mode is a shader that still says what it +/// said before. +/// +/// Five things actually change: +/// +/// The version becomes 450 and the original #extension lines are dropped, +/// since they name GL extensions that either do not exist or are already core in +/// Vulkan GLSL. +/// +/// Loose uniforms move into the program record, set 2's dynamic uniform buffer +/// (GL's default uniform block has no Vulkan equivalent, and this game declares +/// 488 of them), or into the shared frame block at set 0. +/// +/// Every program targets the one shared pipeline layout (plan decision 9, +/// ). A sampler named and typed like one of set 0's +/// frame textures reads that binding. Every other sampler becomes a slot index in +/// the push block under its own name, and every reference to it in the body - a +/// sampling call or an argument to a function - reads +/// optimumTextures<Kind>[name] from set 1. Named uniform blocks become +/// layout(std140) readonly buffer blocks in set 2, so the client's std140 +/// bytes are read unchanged; storage blocks move to set 2 at the convention's +/// bindings. +/// +/// Vertex inputs, varyings and fragment outputs gain the explicit locations +/// SPIR-V requires and GLSL 330 left implicit. +/// +/// The last stage before rasterisation gains a wrapper around main that +/// remaps clip depth from GL's [-w, w] to Vulkan's [0, w]. Nothing else about the +/// coordinate system is touched: no Y flip, no matrix rewriting. GL and Vulkan +/// agree on the relationship between clip space, framebuffer memory and texture +/// coordinates; they disagree only on what to call the origin, and on depth. +/// +internal static class ShaderRewriter +{ + private const string MainReplacementName = "_optimum_main"; + + private readonly record struct Edit(int Start, int Length, string Replacement); + + public static RewrittenShader Rewrite( + ParsedShader parsed, + ProgramInterfaceLayout layout, + EnumShaderType stage, + bool emitDepthRemap) + { + var result = new RewrittenShader(); + string source = parsed.Source; + var edits = new List(); + + AddHeaderEdits(parsed, layout, stage, emitDepthRemap, edits); + + foreach (GlslDeclaration declaration in parsed.Declarations) + { + switch (declaration.Kind) + { + case GlslDeclarationKind.DefaultUniform: + // Its storage now lives in the generated block or the shared + // frame block. Members keep their names in both, so every use + // site still compiles. + if (layout.MembersByName.ContainsKey(declaration.Name) || + layout.FrameMemberDeclaredLengths.ContainsKey(declaration.Name)) + { + edits.Add(new Edit(declaration.Start, declaration.Length, "")); + } + break; + + case GlslDeclarationKind.OpaqueUniform: + if (layout.SamplersByName.TryGetValue(declaration.Name, out SamplerBinding? sampler)) + { + if (sampler.IsFrameTexture) + { + edits.Add(LayoutEdit(declaration, new (string, string?)[] + { + ("set", Number(SetConvention.FrameSet)), + ("binding", Number(sampler.FrameBinding)), + })); + } + else + { + // The name is now the slot index in the push block. + edits.Add(new Edit(declaration.Start, declaration.Length, "")); + } + } + break; + + case GlslDeclarationKind.UniformBlock: + AddBlockEdit(layout.UniformBlocks, declaration, edits, asStorage: true); + break; + + case GlslDeclarationKind.StorageBlock: + AddBlockEdit(layout.StorageBlocks, declaration, edits, asStorage: false); + break; + + case GlslDeclarationKind.Input: + case GlslDeclarationKind.Output: + AddLocationEdit(layout, declaration, stage, edits); + break; + } + } + + AddSamplerReferenceEdits(parsed, layout, stage, edits); + + if (emitDepthRemap) + { + AddDepthRemapEdits(parsed, stage, edits, result); + } + + result.Code = ApplyEdits(source, edits); + return result; + } + + // -------------------------------------------------------------------- header + + private static void AddHeaderEdits( + ParsedShader parsed, ProgramInterfaceLayout layout, EnumShaderType stage, bool emitDepthRemap, + List edits) + { + string frameBlock = BuildFrameBlock(layout, stage); + string block = BuildUniformBlock(layout, stage); + string push = BuildPushBlock(layout, stage); + string arrays = BuildTextureArrays(layout, stage); + + var header = new StringBuilder(); + header.Append("#version 450\n"); + if (frameBlock.Length > 0 || block.Length > 0 || push.Length > 0) + { + header.Append("#extension GL_EXT_scalar_block_layout : require\n"); + } + if (arrays.Length > 0) + { + header.Append("#extension GL_EXT_nonuniform_qualifier : require\n"); + } + header.Append(frameBlock); + header.Append(block); + header.Append(push); + header.Append(arrays); + + // The geometry stage's EmitVertex() replacement lives in the header so + // it precedes every function that may call it. + if (emitDepthRemap && stage == EnumShaderType.GeometryShader) + { + header.Append("void " + EmitVertexReplacementName + "() { " + DepthRemapStatement + " EmitVertex(); }\n"); + } + + if (parsed.VersionStart >= 0) + { + edits.Add(new Edit(parsed.VersionStart, parsed.VersionLength, header.ToString().TrimEnd('\n'))); + } + else + { + edits.Add(new Edit(0, 0, header.ToString())); + } + + // The originals name GL extensions - GL_ARB_explicit_attrib_location and + // friends - that Vulkan GLSL either lacks or already includes. + foreach ((int start, int length) in parsed.ExtensionDirectives) + { + edits.Add(new Edit(start, length, "")); + } + } + + /// + /// Emits the shared frame block () with the members + /// this stage reads, at the offsets every program agrees on, each with the array + /// length this program declared. Anonymous like the program's own block, so + /// every reference in the body resolves unchanged. + /// + private static string BuildFrameBlock(ProgramInterfaceLayout layout, EnumShaderType stage) + { + if (!layout.FrameMembersByStage.TryGetValue(stage, out HashSet? stageMembers)) return ""; + if (stageMembers.Count == 0) return ""; + + var builder = new StringBuilder(); + builder.Append(CultureInfo.InvariantCulture, $"\nlayout(scalar, set = {FrameGlobals.Set}"); + builder.Append(CultureInfo.InvariantCulture, $", binding = {FrameGlobals.Binding}) uniform "); + builder.Append(FrameGlobals.BlockTypeName); + builder.Append("\n{\n"); + + foreach (UniformMember member in FrameGlobals.Members) + { + if (!stageMembers.Contains(member.Name)) continue; + + builder.Append(CultureInfo.InvariantCulture, $" layout(offset = {member.Offset}) "); + builder.Append(member.Type.Name).Append(' ').Append(member.Name); + int length = layout.FrameMemberDeclaredLengths[member.Name]; + if (length > 0) + { + builder.Append('[').Append(length.ToString(CultureInfo.InvariantCulture)).Append(']'); + } + builder.Append(";\n"); + } + + builder.Append("};\n"); + return builder.ToString(); + } + + /// + /// Emits the program record - the block that replaces GL's default uniform + /// block, at set 2's record binding - carrying only the members this stage declared. + /// + /// Members keep their original names and the block is anonymous, so every + /// reference in the shader body resolves unchanged. Each member states its + /// offset explicitly, which is what lets a stage declare a subset without + /// disturbing the shared layout the CPU writes into - and what avoids + /// redefining a name that is a varying in the other stage. + /// + private static string BuildUniformBlock(ProgramInterfaceLayout layout, EnumShaderType stage) + { + if (!layout.HasUniformBlock) return ""; + if (!layout.MembersByStage.TryGetValue(stage, out HashSet? stageMembers)) return ""; + if (stageMembers.Count == 0) return ""; + + var builder = new StringBuilder(); + builder.Append(CultureInfo.InvariantCulture, $"\nlayout(scalar, set = {SetConvention.StorageSet}"); + builder.Append(CultureInfo.InvariantCulture, $", binding = {SetConvention.ProgramRecordBinding}) uniform "); + builder.Append(ProgramInterfaceLayout.BlockTypeName); + builder.Append("\n{\n"); + + foreach (UniformMember member in layout.Members) + { + if (!stageMembers.Contains(member.Name)) continue; + + builder.Append(CultureInfo.InvariantCulture, $" layout(offset = {member.Offset}) "); + builder.Append(member.Type.Name).Append(' ').Append(member.Name); + if (member.ArrayLength > 0) + { + builder.Append('[').Append(member.ArrayLength.ToString(CultureInfo.InvariantCulture)).Append(']'); + } + builder.Append(";\n"); + } + + builder.Append("};\n"); + return builder.ToString(); + } + + /// + /// Emits the push block with one slot index per bindless sampler this stage + /// declared, under the sampler's own name, at the offset the whole program agrees + /// on. Anonymous, so the rewritten references read the index by that name. + /// + private static string BuildPushBlock(ProgramInterfaceLayout layout, EnumShaderType stage) + { + if (!layout.SamplersByStage.TryGetValue(stage, out HashSet? stageSamplers)) return ""; + + var builder = new StringBuilder(); + foreach (SamplerBinding sampler in layout.Samplers) + { + if (sampler.IsFrameTexture || !stageSamplers.Contains(sampler.Name)) continue; + if (builder.Length == 0) + { + builder.Append("\nlayout(push_constant, scalar) uniform ").Append(ProgramInterfaceLayout.PushBlockTypeName); + builder.Append("\n{\n"); + } + // OPTIMUM_SAMPLER_SLOT(, ) in bindings.glsl expands to the same declaration. + builder.Append(CultureInfo.InvariantCulture, $" layout(offset = {sampler.PushOffset}) uint {sampler.Name};"); + builder.Append(CultureInfo.InvariantCulture, $" // {sampler.TypeName}\n"); + } + if (builder.Length == 0) return ""; + builder.Append("};\n"); + return builder.ToString(); + } + + /// The set 1 arrays this stage's bindless samplers index, declared as bindings.glsl declares them. + private static string BuildTextureArrays(ProgramInterfaceLayout layout, EnumShaderType stage) + { + if (!layout.SamplersByStage.TryGetValue(stage, out HashSet? stageSamplers)) return ""; + + var kinds = new SortedSet(); + foreach (SamplerBinding sampler in layout.Samplers) + { + if (!sampler.IsFrameTexture && stageSamplers.Contains(sampler.Name)) kinds.Add((int)sampler.Kind); + } + var builder = new StringBuilder(); + foreach (int kind in kinds) + { + SetConvention.Binding array = SetConvention.TextureArrays[kind]; + builder.Append(CultureInfo.InvariantCulture, + $"layout(set = {SetConvention.TextureSet}, binding = {array.Value}) uniform {array.GlslType} {array.Name}[];\n"); + } + return builder.ToString(); + } + + // -------------------------------------------------------------- declarations + + private static readonly string[] MemoryLayouts = { "std140", "std430", "shared", "packed" }; + + /// + /// Moves a named block to its set 2 binding. A uniform block becomes a + /// layout(std140) readonly buffer: std140 is what the client's UBO uploads + /// already stride to, and a storage buffer defaults to std430, so the memory + /// layout is stated explicitly whatever the shader wrote. + /// + private static void AddBlockEdit(List blocks, GlslDeclaration declaration, List edits, + bool asStorage) + { + foreach (BlockBinding block in blocks) + { + if (block.BlockName != declaration.Name) continue; + + if (asStorage) + { + edits.Add(LayoutEdit(declaration, new (string, string?)[] + { + ("std140", null), + ("set", Number(block.Set)), + ("binding", Number(block.Binding)), + }, MemoryLayouts)); + if (declaration.StorageKeywordStart >= 0) + { + edits.Add(new Edit(declaration.StorageKeywordStart, "uniform".Length, "readonly buffer")); + } + return; + } + + // A storage block keeps the memory layout it chose (std430 for faceDataBuf). + edits.Add(LayoutEdit(declaration, new (string, string?)[] + { + ("set", Number(block.Set)), + ("binding", Number(block.Binding)), + })); + return; + } + } + + /// + /// Rewrites every reference to a bindless sampler this stage declared into + /// optimumTextures<Kind>[name], where name is now the slot index + /// in the push block. A sampler can only ever appear as a function argument - to + /// texture, texelFetch, textureLod, textureGather, + /// textureSize or to a function of the shader's own such as colormap's + /// getColorMapped - so every reference is rewritten and no call form is + /// singled out. + /// + /// Not rewritten: the global declarations (they have edits of their own), a field + /// after a dot, a declaration of the same name (a parameter sampler2D tex, a + /// local or struct member), and every use inside the scope such a declaration + /// shadows the global in. Comments and preprocessor lines are skipped. + /// + private static void AddSamplerReferenceEdits( + ParsedShader parsed, ProgramInterfaceLayout layout, EnumShaderType stage, List edits) + { + if (!layout.SamplersByStage.TryGetValue(stage, out HashSet? stageSamplers)) return; + + var replacements = new Dictionary(StringComparer.Ordinal); + foreach (SamplerBinding sampler in layout.Samplers) + { + if (sampler.IsFrameTexture || !stageSamplers.Contains(sampler.Name)) continue; + replacements[sampler.Name] = SetConvention.TextureArrays[(int)sampler.Kind].Name + "[" + sampler.Name + "]"; + } + if (replacements.Count == 0) return; + + var skipped = new List<(int Start, int End)>(); + foreach (GlslDeclaration declaration in parsed.Declarations) + { + if (declaration.Kind is GlslDeclarationKind.Other) continue; + skipped.Add((declaration.Start, declaration.End)); + } + skipped.Sort(static (a, b) => a.Start.CompareTo(b.Start)); + + string source = parsed.Source; + var scopes = new List> { new(StringComparer.Ordinal) }; + HashSet? parameters = null; + int parenDepth = 0; + string previous = ";"; + bool previousIsWord = false; + int skip = 0; + int i = 0; + bool lineStart = true; + + while (i < source.Length) + { + while (skip < skipped.Count && skipped[skip].End <= i) skip++; + if (skip < skipped.Count && skipped[skip].Start <= i) + { + i = skipped[skip].End; + previous = ";"; + previousIsWord = false; + continue; + } + + char c = source[i]; + if (c == '\n') { lineStart = true; i++; continue; } + if (char.IsWhiteSpace(c)) { i++; continue; } + + if (c == '#' && lineStart) + { + while (i < source.Length && source[i] != '\n') + { + if (source[i] == '\\' && i + 1 < source.Length && source[i + 1] == '\n') i++; + i++; + } + continue; + } + lineStart = false; + + if (c == '/' && i + 1 < source.Length && source[i + 1] == '/') + { + while (i < source.Length && source[i] != '\n') i++; + continue; + } + if (c == '/' && i + 1 < source.Length && source[i + 1] == '*') + { + int close = source.IndexOf("*/", i + 2, StringComparison.Ordinal); + i = close < 0 ? source.Length : close + 2; + continue; + } + + if (char.IsLetter(c) || c == '_') + { + int start = i; + while (i < source.Length && (char.IsLetterOrDigit(source[i]) || source[i] == '_')) i++; + string word = source.Substring(start, i - start); + + if (replacements.TryGetValue(word, out string? replacement) && previous != ".") + { + bool declaration = previousIsWord && previous is not ("return" or "case"); + if (declaration) + { + (scopes.Count == 1 && parenDepth > 0 ? parameters ??= new(StringComparer.Ordinal) : scopes[^1]) + .Add(word); + } + else if (!Shadowed(scopes, word)) + { + edits.Add(new Edit(start, word.Length, replacement)); + } + } + + previous = word; + previousIsWord = true; + continue; + } + + switch (c) + { + case '(': + if (scopes.Count == 1 && parenDepth == 0) parameters = null; + parenDepth++; + break; + case ')': + if (parenDepth > 0) parenDepth--; + break; + case '{': + // A function body sees its parameters; any other brace opens a plain scope. + scopes.Add(scopes.Count == 1 && parameters != null ? parameters : new(StringComparer.Ordinal)); + parameters = null; + break; + case '}': + if (scopes.Count > 1) scopes.RemoveAt(scopes.Count - 1); + break; + case ';': + if (scopes.Count == 1) parameters = null; + break; + } + previous = c.ToString(); + previousIsWord = false; + i++; + } + } + + private static bool Shadowed(List> scopes, string name) + { + // The outermost set only ever holds names declared at global scope inside a + // struct or prototype parenthesis, which never shadow a use; start above it. + for (int i = scopes.Count - 1; i >= 1; i--) + { + if (scopes[i].Contains(name)) return true; + } + return false; + } + + private static void AddLocationEdit( + ProgramInterfaceLayout layout, GlslDeclaration declaration, EnumShaderType stage, List edits) + { + int location; + if (stage == EnumShaderType.VertexShader && declaration.Kind == GlslDeclarationKind.Input) + { + if (!layout.VertexInputLocations.TryGetValue(declaration.Name, out location)) return; + } + else if (stage == EnumShaderType.FragmentShader && declaration.Kind == GlslDeclarationKind.Output) + { + if (!layout.FragmentOutputLocations.TryGetValue(declaration.Name, out location)) return; + } + else + { + if (!layout.VaryingLocations.TryGetValue(declaration.Name, out location)) return; + } + + edits.Add(LayoutEdit(declaration, new (string, string?)[] + { + ("location", Number(location)), + })); + } + + private static string Number(int value) => value.ToString(CultureInfo.InvariantCulture); + + /// + /// Produces an edit that replaces the declaration's layout(...) clause + /// with one carrying the given keys (a null value adds a bare word such as + /// std140), preserving any others it already had except + /// . When there was no clause, the span is empty + /// and this inserts one. + /// + private static Edit LayoutEdit(GlslDeclaration declaration, (string Key, string? Value)[] additions, + params string[] removals) + { + var parts = new List(); + var overridden = new HashSet(removals, StringComparer.Ordinal); + foreach ((string key, _) in additions) overridden.Add(key); + + if (declaration.LayoutQualifiers != null) + { + foreach (string raw in declaration.LayoutQualifiers.Split(',')) + { + string part = raw.Trim(); + if (part.Length == 0) continue; + + int equals = part.IndexOf('='); + string key = (equals < 0 ? part : part[..equals]).Trim(); + if (overridden.Contains(key)) continue; + + parts.Add(part); + } + } + + foreach ((string key, string? value) in additions) + { + parts.Add(value == null ? key : $"{key} = {value}"); + } + + // A replaced clause keeps the whitespace that followed it; an inserted one brings its own. + string clause = $"layout({string.Join(", ", parts)})"; + return new Edit(declaration.LayoutStart, declaration.LayoutLength, + declaration.LayoutLength == 0 ? clause + " " : clause); + } + + // ---------------------------------------------------------------- depth remap + + /// + /// Wraps main so clip-space depth lands in Vulkan's [0, w] range. + /// + /// Doing it here rather than by folding a correction into the projection + /// matrix keeps every matrix in the game untouched - the frustum culler, the + /// shadow orthographic projections and any matrix a mod builds all keep + /// working, and the CPU-side code never has to know which backend is running. + /// + private const string DepthRemapStatement = "gl_Position.z = (gl_Position.z + gl_Position.w) * 0.5;"; + + private static void AddDepthRemapEdits( + ParsedShader parsed, EnumShaderType stage, List edits, RewrittenShader result) + { + if (stage == EnumShaderType.GeometryShader) + { + AddGeometryDepthRemapEdits(parsed, edits, result); + return; + } + + if (!parsed.HasMain) + { + result.Errors.Add("stage has no main() to wrap for the Vulkan depth range"); + return; + } + + edits.Add(new Edit(parsed.MainNameStart, "main".Length, MainReplacementName)); + + edits.Add(new Edit(parsed.Source.Length, 0, + "\n\nvoid main()\n{\n" + + " " + MainReplacementName + "();\n" + + " " + DepthRemapStatement + "\n" + + "}\n")); + } + + private const string EmitVertexReplacementName = "_optimum_emit_vertex"; + + /// + /// A geometry stage snapshots gl_Position at every EmitVertex(), + /// so a wrapper around main would run after every vertex has already + /// left. Each call is redirected to a helper that remaps and then emits, + /// which keeps the call a single statement: an unbraced if or loop + /// body around it keeps its scope, where an inserted extra statement would + /// not. + /// + private static void AddGeometryDepthRemapEdits(ParsedShader parsed, List edits, RewrittenShader result) + { + string source = parsed.Source; + const string call = "EmitVertex"; + int found = 0; + + for (int at = source.IndexOf(call, StringComparison.Ordinal); at >= 0; + at = source.IndexOf(call, at + call.Length, StringComparison.Ordinal)) + { + bool startsWord = at == 0 || !(char.IsLetterOrDigit(source[at - 1]) || source[at - 1] == '_'); + int after = at + call.Length; + while (after < source.Length && char.IsWhiteSpace(source[after])) after++; + bool isCall = after < source.Length && source[after] == '('; + if (!startsWord || !isCall) continue; + + edits.Add(new Edit(at, call.Length, EmitVertexReplacementName)); + found++; + } + + if (found == 0) + { + result.Errors.Add("geometry stage never calls EmitVertex(), so no vertex gets the Vulkan depth range"); + } + } + + // --------------------------------------------------------------------- edits + + /// + /// Applies edits back to front so earlier offsets stay valid. Overlapping + /// edits are a programming error here, not a shader error, so they assert + /// rather than being silently resolved. + /// + private static string ApplyEdits(string source, List edits) + { + edits.Sort(static (a, b) => b.Start != a.Start ? b.Start.CompareTo(a.Start) : b.Length.CompareTo(a.Length)); + + var builder = new StringBuilder(source); + int previousStart = int.MaxValue; + + foreach (Edit edit in edits) + { + if (edit.Start + edit.Length > previousStart) + { + throw new InvalidOperationException( + $"overlapping shader edits at {edit.Start}..{edit.Start + edit.Length} and {previousStart}"); + } + builder.Remove(edit.Start, edit.Length); + builder.Insert(edit.Start, edit.Replacement); + previousStart = edit.Start; + } + + return builder.ToString(); + } +} diff --git a/Optimum.Render.Vulkan/Shaders/ShaderTranslator.cs b/Optimum.Render.Vulkan/Shaders/ShaderTranslator.cs new file mode 100644 index 00000000..10297f4a --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/ShaderTranslator.cs @@ -0,0 +1,162 @@ +using System; +using System.Collections.Generic; +using Vintagestory.API.Client; + +namespace Optimum.Render.Vulkan.Shaders; + +/// One stage's input to translation. +internal sealed class ShaderStageSource +{ + public EnumShaderType Stage; + /// Include-expanded GLSL, as ShaderRegistry produces it. + public string Code = ""; + /// The program's #define block. + public string PrefixCode = ""; + public string Filename = "shader"; +} + +/// A whole program, translated and ready to become pipeline stages. +internal sealed class TranslatedProgram +{ + public ProgramInterfaceLayout Layout = new(); + public Dictionary Spirv { get; } = new(); + /// The rewritten GLSL per stage, kept for diagnostics. + public Dictionary RewrittenSource { get; } = new(); + public List Errors { get; } = new(); + public bool Success => Errors.Count == 0; + + /// + /// The specialization constants a native program's pipelines are created with + /// (docs/vulkan.md); null for a rewritten program, whose + /// defines were resolved by the preprocessor. + /// + public NativeSpecialization? Specialization; + + /// Whether the program was linked from the manifest's SPIR-V rather than through the rewriter. + public bool IsNative; +} + +/// +/// Drives a program from GLSL 330 to SPIR-V. +/// +/// The order is forced by GL's semantics rather than chosen: preprocess every +/// stage first (declarations hide behind #if), then parse them all, then +/// resolve the program-wide interface, and only then rewrite and compile. Nothing +/// can be finalised per stage because uniforms and varyings are matched by name +/// across the whole program. +/// +internal static class ShaderTranslator +{ + /// + /// Stages in the order the interface layout walks them. Fixed rather than + /// incidental, so uniform offsets and varying locations - and therefore the + /// SPIR-V cache key - are identical from run to run. + /// + private static readonly EnumShaderType[] StageOrder = + { + EnumShaderType.VertexShader, + EnumShaderType.FragmentShader, + EnumShaderType.GeometryShader, + }; + + /// + /// The include files the program was assembled from, as ShaderRegistry records + /// them. They decide which uniforms read the shared frame block + /// (); without them every uniform stays the program's own. + /// + public static TranslatedProgram Translate( + IReadOnlyList stages, + ShaderCompiler compiler, + IReadOnlyDictionary? declaredAttributes = null, + IReadOnlySet? includes = null) + { + var program = new TranslatedProgram(); + + var ordered = new List(); + foreach (EnumShaderType stage in StageOrder) + { + foreach (ShaderStageSource candidate in stages) + { + if (candidate.Stage == stage) ordered.Add(candidate); + } + } + + // Preprocess and parse. + var parsed = new List<(EnumShaderType Stage, ParsedShader Parsed)>(); + foreach (ShaderStageSource stage in ordered) + { + ShaderCompileResult preprocessed = + compiler.Preprocess(stage.Code, stage.PrefixCode, stage.Filename, stage.Stage); + + if (!preprocessed.Success) + { + program.Errors.Add($"{stage.Filename}: preprocessing failed: {preprocessed.Error}"); + continue; + } + + // Rename identifiers 4.50 reserved before anything reads the source, + // so the parser and the rewriter both see the same names. + string source = GlslReservedWords.Rename(preprocessed.PreprocessedText); + parsed.Add((stage.Stage, GlslParser.Parse(source))); + } + + if (program.Errors.Count > 0) return program; + + if (parsed.Count == 0) + { + program.Errors.Add("no shader stage survived translation"); + return program; + } + + program.Layout = ProgramInterfaceLayout.Build(parsed, declaredAttributes, includes); + foreach (string error in program.Layout.Errors) + { + program.Errors.Add(error); + } + if (program.Errors.Count > 0) return program; + + // The depth remap belongs on the last stage before rasterisation. + EnumShaderType depthRemapStage = EnumShaderType.VertexShader; + foreach ((EnumShaderType stage, _) in parsed) + { + if (stage == EnumShaderType.GeometryShader) depthRemapStage = stage; + } + + // Rewrite and compile. + foreach ((EnumShaderType stage, ParsedShader shader) in parsed) + { + string filename = FilenameFor(ordered, stage); + + RewrittenShader rewritten = + ShaderRewriter.Rewrite(shader, program.Layout, stage, emitDepthRemap: stage == depthRemapStage); + + program.RewrittenSource[stage] = rewritten.Code; + + foreach (string error in rewritten.Errors) + { + program.Errors.Add($"{filename}: {error}"); + } + if (rewritten.HasErrors) continue; + + ShaderCompileResult compiled = compiler.Compile(rewritten.Code, filename, stage); + if (!compiled.Success) + { + program.Errors.Add($"{filename}: {compiled.Error}"); + continue; + } + + program.Spirv[stage] = compiled.Spirv; + } + + return program; + } + + private static string FilenameFor(IReadOnlyList stages, EnumShaderType stage) + { + foreach (ShaderStageSource candidate in stages) + { + if (candidate.Stage == stage) return candidate.Filename; + } + return stage.ToString(); + } +} diff --git a/Optimum.Render.Vulkan/Shaders/SpirvReflection.cs b/Optimum.Render.Vulkan/Shaders/SpirvReflection.cs new file mode 100644 index 00000000..a27d4a6c --- /dev/null +++ b/Optimum.Render.Vulkan/Shaders/SpirvReflection.cs @@ -0,0 +1,822 @@ +using System; +using System.Collections.Generic; +using System.Text; + +namespace Optimum.Render.Vulkan.Shaders; + +/// What a descriptor binding holds, as the SPIR-V declares it. +internal enum SpirvDescriptorKind +{ + CombinedImageSampler, + SampledImage, + StorageImage, + Sampler, + UniformBuffer, + StorageBuffer, +} + +/// A member of a uniform, push-constant or storage block. +internal sealed class SpirvBlockMember +{ + public string Name = ""; + /// The GLSL spelling of the element type (mat4, vec3, a struct's type name). + public string GlslType = ""; + public int Offset; + /// Bytes the member occupies: element size, or length times the array stride. + public int Size; + /// 0 when not an array, -1 for a runtime-sized array. + public int ArrayLength; +} + +/// A struct used as a block, with explicit member offsets. +internal sealed class SpirvBlock +{ + public string TypeName = ""; + public string InstanceName = ""; + /// End of the last member, which is what a scalar-layout block needs in bytes. + public int Size; + public List Members = new(); +} + +internal sealed class SpirvDescriptorBinding +{ + public uint VariableId; + public string Name = ""; + public int Set; + public int Binding; + public SpirvDescriptorKind Kind; + /// For images and samplers the GLSL type (sampler2DArrayShadow); for blocks the block type name. + public string GlslType = ""; + /// 0 when not an array. + public int ArrayLength; + public bool RuntimeArray; + /// The block layout for uniform and storage buffers, null otherwise. + public SpirvBlock? Block; +} + +internal sealed class SpirvInterfaceVariable +{ + public uint VariableId; + public string Name = ""; + public int Location; + public string GlslType = ""; + /// 0 when not an array. + public int ArrayLength; +} + +internal sealed class SpirvSpecConstant +{ + public int SpecId; + public string Name = ""; + /// bool, int, uint, float or double. + public string GlslType = ""; + /// The default as a number; a bool is 0 or 1. + public double DefaultValue; +} + +/// Everything the native shader manifest needs from one SPIR-V module. +internal sealed class SpirvModuleReflection +{ + public string EntryPoint = ""; + /// The SPIR-V execution model: 0 vertex, 3 geometry, 4 fragment, 5 compute. + public int ExecutionModel; + public List Inputs = new(); + public List Outputs = new(); + public List Bindings = new(); + public SpirvBlock? PushConstants; + public List SpecConstants = new(); + /// Descriptor, push and interface variables some function body actually uses. + public HashSet UsedVariables = new(); + /// Locations of the outputs a function body stores to (array outputs contribute every element). + public SortedSet WrittenOutputLocations = new(); + /// + /// For every push-constant member whose value is used as the first index into an arrayed + /// descriptor: the (set, binding) pairs it indexes. This is how a sampler slot is tied to the + /// bindless array it selects from, independent of names (an optimised module has none). + /// + public Dictionary> PushMemberIndexes = new(); +} + +/// +/// A small SPIR-V reader for the offline shader compiler and the native program manifest +/// (docs/vulkan.md). It reads entry points, names, decorations and the +/// type graph; it does not validate the module, which shaderc produced moments earlier. +/// +/// An optimised module has no OpNames and has dropped whatever nothing uses, so the +/// builder reflects two modules per stage: the unoptimised one for declarations and names, +/// the shipped one for use (, written outputs, +/// slot dataflow). +/// +internal static class SpirvReflection +{ + public const uint Magic = 0x07230203; + + public const int ExecutionModelVertex = 0; + public const int ExecutionModelGeometry = 3; + public const int ExecutionModelFragment = 4; + + // Opcodes. + private const int OpName = 5; + private const int OpMemberName = 6; + private const int OpEntryPoint = 15; + private const int OpTypeVoid = 19; + private const int OpTypeBool = 20; + private const int OpTypeInt = 21; + private const int OpTypeFloat = 22; + private const int OpTypeVector = 23; + private const int OpTypeMatrix = 24; + private const int OpTypeImage = 25; + private const int OpTypeSampler = 26; + private const int OpTypeSampledImage = 27; + private const int OpTypeArray = 28; + private const int OpTypeRuntimeArray = 29; + private const int OpTypeStruct = 30; + private const int OpTypePointer = 32; + private const int OpConstantTrue = 41; + private const int OpConstantFalse = 42; + private const int OpConstant = 43; + private const int OpSpecConstantTrue = 48; + private const int OpSpecConstantFalse = 49; + private const int OpSpecConstant = 50; + private const int OpFunction = 54; + private const int OpFunctionCall = 57; + private const int OpVariable = 59; + private const int OpImageTexelPointer = 60; + private const int OpLoad = 61; + private const int OpStore = 62; + private const int OpCopyMemory = 63; + private const int OpAccessChain = 65; + private const int OpInBoundsAccessChain = 66; + private const int OpPtrAccessChain = 67; + private const int OpArrayLength = 68; + private const int OpCompositeExtract = 81; + private const int OpCopyObject = 83; + private const int OpUConvert = 113; + private const int OpSConvert = 114; + private const int OpBitcast = 124; + + // Decorations. + private const int DecorationSpecId = 1; + private const int DecorationBlock = 2; + private const int DecorationBufferBlock = 3; + private const int DecorationArrayStride = 6; + private const int DecorationMatrixStride = 7; + private const int DecorationBuiltIn = 11; + private const int DecorationLocation = 30; + private const int DecorationBinding = 33; + private const int DecorationDescriptorSet = 34; + private const int DecorationOffset = 35; + + // Storage classes. + private const int StorageUniformConstant = 0; + private const int StorageInput = 1; + private const int StorageUniform = 2; + private const int StorageOutput = 3; + private const int StoragePushConstant = 9; + private const int StorageStorageBuffer = 12; + + private sealed class TypeInfo + { + public int Op; + public uint[] Operands = Array.Empty(); + } + + private sealed class Module + { + public readonly Dictionary Names = new(); + public readonly Dictionary<(uint, int), string> MemberNames = new(); + public readonly Dictionary> Decorations = new(); + public readonly Dictionary<(uint, int), List<(int Decoration, uint[] Literals)>> MemberDecorations = new(); + public readonly Dictionary Types = new(); + public readonly Dictionary Constants = new(); + public readonly List<(uint Id, uint Type, int Op, uint[] Literals)> SpecConstants = new(); + public readonly Dictionary Variables = new(); + public readonly List<(int Op, uint[] Operands)> Body = new(); + public string EntryName = ""; + public int ExecutionModel = -1; + public uint[] Interface = Array.Empty(); + } + + public static SpirvModuleReflection Reflect(byte[] spirv) + { + if (spirv == null || spirv.Length < 20 || spirv.Length % 4 != 0) + { + throw new FormatException("not a SPIR-V module: length " + (spirv?.Length ?? 0)); + } + + var words = new uint[spirv.Length / 4]; + Buffer.BlockCopy(spirv, 0, words, 0, spirv.Length); + if (words[0] != Magic) + { + throw new FormatException("not a SPIR-V module: magic 0x" + words[0].ToString("x8")); + } + + Module module = Parse(words); + return Build(module); + } + + private static Module Parse(uint[] words) + { + var module = new Module(); + bool inFunctions = false; + int position = 5; + while (position < words.Length) + { + int count = (int)(words[position] >> 16); + int op = (int)(words[position] & 0xFFFF); + if (count == 0 || position + count > words.Length) + { + throw new FormatException("truncated SPIR-V instruction at word " + position); + } + var operands = new uint[count - 1]; + Array.Copy(words, position + 1, operands, 0, count - 1); + position += count; + + if (op == OpFunction) inFunctions = true; + if (inFunctions) + { + module.Body.Add((op, operands)); + continue; + } + + switch (op) + { + case OpName: + module.Names[operands[0]] = ReadString(operands, 1, out _); + break; + case OpMemberName: + module.MemberNames[(operands[0], (int)operands[1])] = ReadString(operands, 2, out _); + break; + case OpEntryPoint: + if (module.ExecutionModel < 0) + { + module.ExecutionModel = (int)operands[0]; + module.EntryName = ReadString(operands, 2, out int next); + module.Interface = operands[next..]; + } + break; + case 71: // OpDecorate + Add(module.Decorations, operands[0], ((int)operands[1], operands[2..])); + break; + case 72: // OpMemberDecorate + Add(module.MemberDecorations, (operands[0], (int)operands[1]), ((int)operands[2], operands[3..])); + break; + case OpTypeVoid: + case OpTypeBool: + case OpTypeInt: + case OpTypeFloat: + case OpTypeVector: + case OpTypeMatrix: + case OpTypeImage: + case OpTypeSampler: + case OpTypeSampledImage: + case OpTypeArray: + case OpTypeRuntimeArray: + case OpTypeStruct: + module.Types[operands[0]] = new TypeInfo { Op = op, Operands = operands[1..] }; + break; + case OpTypePointer: + module.Types[operands[0]] = new TypeInfo { Op = op, Operands = operands[1..] }; + break; + case OpConstant: + module.Constants[operands[1]] = (operands[0], operands[2..]); + break; + case OpConstantTrue: + case OpConstantFalse: + module.Constants[operands[1]] = (operands[0], new uint[] { op == OpConstantTrue ? 1u : 0u }); + break; + case OpSpecConstant: + case OpSpecConstantTrue: + case OpSpecConstantFalse: + module.SpecConstants.Add((operands[1], operands[0], op, operands[2..])); + break; + case OpVariable: + module.Variables[operands[1]] = (operands[0], (int)operands[2]); + break; + } + } + return module; + } + + private static SpirvModuleReflection Build(Module module) + { + var result = new SpirvModuleReflection + { + EntryPoint = module.EntryName, + ExecutionModel = module.ExecutionModel, + }; + + uint? pushVariable = null; + foreach ((uint id, (uint pointerType, int storageClass)) in module.Variables) + { + uint pointee = module.Types.TryGetValue(pointerType, out TypeInfo? pointer) && pointer.Op == OpTypePointer + ? pointer.Operands[1] + : 0; + if (pointee == 0) continue; + + switch (storageClass) + { + case StorageInput: + case StorageOutput: + { + if (HasDecoration(module, id, DecorationBuiltIn) || IsBuiltInBlock(module, pointee)) continue; + if (!TryDecoration(module, id, DecorationLocation, out uint location)) continue; + UnwrapArray(module, pointee, out uint element, out int length, out _); + var variable = new SpirvInterfaceVariable + { + VariableId = id, + Name = NameOf(module, id), + Location = (int)location, + GlslType = GlslTypeName(module, element), + ArrayLength = length, + }; + (storageClass == StorageInput ? result.Inputs : result.Outputs).Add(variable); + break; + } + case StoragePushConstant: + pushVariable = id; + result.PushConstants = ReadBlock(module, pointee, id); + break; + case StorageUniformConstant: + case StorageUniform: + case StorageStorageBuffer: + { + if (!TryDecoration(module, id, DecorationDescriptorSet, out uint set) || + !TryDecoration(module, id, DecorationBinding, out uint binding)) + { + continue; + } + UnwrapArray(module, pointee, out uint element, out int length, out bool runtime); + var descriptor = new SpirvDescriptorBinding + { + VariableId = id, + Name = NameOf(module, id), + Set = (int)set, + Binding = (int)binding, + ArrayLength = length, + RuntimeArray = runtime, + }; + TypeInfo elementType = module.Types[element]; + if (elementType.Op == OpTypeStruct) + { + bool bufferBlock = HasDecoration(module, element, DecorationBufferBlock); + descriptor.Kind = storageClass == StorageStorageBuffer || bufferBlock + ? SpirvDescriptorKind.StorageBuffer + : SpirvDescriptorKind.UniformBuffer; + descriptor.Block = ReadBlock(module, element, id); + descriptor.GlslType = descriptor.Block.TypeName; + } + else + { + descriptor.Kind = elementType.Op switch + { + OpTypeSampledImage => SpirvDescriptorKind.CombinedImageSampler, + OpTypeSampler => SpirvDescriptorKind.Sampler, + OpTypeImage when elementType.Operands[5] == 2 => SpirvDescriptorKind.StorageImage, + _ => SpirvDescriptorKind.SampledImage, + }; + descriptor.GlslType = GlslTypeName(module, element); + } + result.Bindings.Add(descriptor); + break; + } + } + } + + result.Inputs.Sort((a, b) => a.Location.CompareTo(b.Location)); + result.Outputs.Sort((a, b) => a.Location.CompareTo(b.Location)); + result.Bindings.Sort((a, b) => a.Set != b.Set ? a.Set.CompareTo(b.Set) : a.Binding.CompareTo(b.Binding)); + + foreach ((uint id, uint type, int op, uint[] literals) in module.SpecConstants) + { + if (!TryDecoration(module, id, DecorationSpecId, out uint specId)) continue; + string typeName = GlslTypeName(module, type); + result.SpecConstants.Add(new SpirvSpecConstant + { + SpecId = (int)specId, + Name = NameOf(module, id), + GlslType = typeName, + DefaultValue = op switch + { + OpSpecConstantTrue => 1, + OpSpecConstantFalse => 0, + _ => DecodeScalar(module, type, literals), + }, + }); + } + result.SpecConstants.Sort((a, b) => a.SpecId.CompareTo(b.SpecId)); + + AnalyseBodies(module, result, pushVariable); + return result; + } + + /// + /// One pass over the function bodies: which variables are used, which outputs are stored to, + /// and which push members index which arrayed descriptors. + /// + private static void AnalyseBodies(Module module, SpirvModuleReflection result, uint? pushVariable) + { + // Pointer results rooted at a global variable. + var roots = new Dictionary(); + // Pointer results that address one top-level member of the push block. + var pushMemberPointers = new Dictionary(); + // Values that carry a push member's value unchanged (or through an integer cast). + var pushMemberValues = new Dictionary(); + // Values that hold the whole push block. + var pushStructValues = new HashSet(); + + uint Root(uint id) => roots.TryGetValue(id, out uint root) ? root : id; + + void Use(uint pointer) + { + uint root = Root(pointer); + if (module.Variables.ContainsKey(root)) result.UsedVariables.Add(root); + } + + void Write(uint pointer) + { + uint root = Root(pointer); + if (!module.Variables.TryGetValue(root, out (uint PointerType, int StorageClass) variable) || + variable.StorageClass != StorageOutput) + { + return; + } + if (!TryDecoration(module, root, DecorationLocation, out uint location)) return; + uint pointee = module.Types[variable.PointerType].Operands[1]; + UnwrapArray(module, pointee, out _, out int length, out _); + for (int i = 0; i < Math.Max(1, length); i++) result.WrittenOutputLocations.Add((int)location + i); + } + + foreach ((int op, uint[] operands) in module.Body) + { + switch (op) + { + case OpAccessChain: + case OpInBoundsAccessChain: + case OpPtrAccessChain: + { + uint id = operands[1]; + uint baseId = operands[2]; + roots[id] = Root(baseId); + Use(baseId); + + int firstIndex = op == OpPtrAccessChain ? 4 : 3; + if (pushVariable.HasValue && baseId == pushVariable.Value && operands.Length == firstIndex + 1 && + TryConstant(module, operands[firstIndex], out long member)) + { + pushMemberPointers[id] = (int)member; + } + + if (operands.Length > firstIndex && + module.Variables.TryGetValue(baseId, out (uint PointerType, int StorageClass) arrayVariable) && + arrayVariable.StorageClass is StorageUniformConstant or StorageUniform or StorageStorageBuffer && + pushMemberValues.TryGetValue(operands[firstIndex], out int slotMember) && + TryDecoration(module, baseId, DecorationDescriptorSet, out uint set) && + TryDecoration(module, baseId, DecorationBinding, out uint binding)) + { + if (!result.PushMemberIndexes.TryGetValue(slotMember, out SortedSet<(int, int)>? indexed)) + { + indexed = new SortedSet<(int, int)>(); + result.PushMemberIndexes[slotMember] = indexed; + } + indexed.Add(((int)set, (int)binding)); + } + break; + } + case OpLoad: + { + uint id = operands[1]; + uint pointer = operands[2]; + Use(pointer); + if (pushMemberPointers.TryGetValue(pointer, out int member)) pushMemberValues[id] = member; + if (pushVariable.HasValue && pointer == pushVariable.Value) pushStructValues.Add(id); + break; + } + case OpStore: + Use(operands[0]); + Write(operands[0]); + break; + case OpCopyMemory: + Use(operands[0]); + Use(operands[1]); + Write(operands[0]); + break; + case OpImageTexelPointer: + case OpArrayLength: + Use(operands[2]); + break; + case OpFunctionCall: + for (int i = 3; i < operands.Length; i++) Use(operands[i]); + break; + case OpCompositeExtract: + if (pushStructValues.Contains(operands[2]) && operands.Length == 4) + { + pushMemberValues[operands[1]] = (int)operands[3]; + } + break; + case OpCopyObject: + case OpUConvert: + case OpSConvert: + case OpBitcast: + if (pushMemberValues.TryGetValue(operands[2], out int carried)) pushMemberValues[operands[1]] = carried; + if (roots.ContainsKey(operands[2]) || module.Variables.ContainsKey(operands[2])) + { + roots[operands[1]] = Root(operands[2]); + } + break; + } + } + } + + private static SpirvBlock ReadBlock(Module module, uint structType, uint variable) + { + var block = new SpirvBlock + { + TypeName = NameOf(module, structType), + InstanceName = NameOf(module, variable), + }; + TypeInfo type = module.Types[structType]; + if (type.Op != OpTypeStruct) return block; + + int end = 0; + for (int index = 0; index < type.Operands.Length; index++) + { + uint memberType = type.Operands[index]; + UnwrapArray(module, memberType, out uint element, out int length, out bool runtime); + TryMemberDecoration(module, structType, index, DecorationOffset, out uint offset); + TryMemberDecoration(module, structType, index, DecorationMatrixStride, out uint matrixStride); + + int elementSize = SizeOf(module, element, (int)matrixStride); + int size; + if (runtime) + { + size = 0; + } + else if (length > 0) + { + size = TryDecoration(module, memberType, DecorationArrayStride, out uint stride) + ? (int)stride * length + : elementSize * length; + } + else + { + size = elementSize; + } + + block.Members.Add(new SpirvBlockMember + { + Name = module.MemberNames.TryGetValue((structType, index), out string? name) ? name : "", + GlslType = GlslTypeName(module, element), + Offset = (int)offset, + Size = size, + ArrayLength = runtime ? -1 : length, + }); + end = Math.Max(end, (int)offset + size); + } + block.Size = end; + return block; + } + + /// Size of a non-array type under its explicit layout; a matrix uses its member's stride. + private static int SizeOf(Module module, uint typeId, int matrixStride) + { + TypeInfo type = module.Types[typeId]; + switch (type.Op) + { + case OpTypeBool: + return 4; + case OpTypeInt: + case OpTypeFloat: + return (int)type.Operands[0] / 8; + case OpTypeVector: + return SizeOf(module, type.Operands[0], 0) * (int)type.Operands[1]; + case OpTypeMatrix: + { + int columns = (int)type.Operands[1]; + int columnSize = SizeOf(module, type.Operands[0], 0); + return matrixStride > 0 ? (columns - 1) * matrixStride + Math.Max(columnSize, matrixStride) : columns * columnSize; + } + case OpTypeArray: + { + UnwrapArray(module, typeId, out uint element, out int length, out _); + return TryDecoration(module, typeId, DecorationArrayStride, out uint stride) + ? (int)stride * length + : SizeOf(module, element, 0) * length; + } + case OpTypeStruct: + { + int end = 0; + for (int index = 0; index < type.Operands.Length; index++) + { + TryMemberDecoration(module, typeId, index, DecorationOffset, out uint offset); + TryMemberDecoration(module, typeId, index, DecorationMatrixStride, out uint stride); + end = Math.Max(end, (int)offset + SizeOf(module, type.Operands[index], (int)stride)); + } + return end; + } + default: + return 0; + } + } + + private static void UnwrapArray(Module module, uint typeId, out uint element, out int length, out bool runtime) + { + element = typeId; + length = 0; + runtime = false; + TypeInfo type = module.Types[typeId]; + if (type.Op == OpTypeRuntimeArray) + { + element = type.Operands[0]; + runtime = true; + } + else if (type.Op == OpTypeArray) + { + element = type.Operands[0]; + length = TryConstant(module, type.Operands[1], out long value) ? (int)value : 0; + } + } + + private static bool IsBuiltInBlock(Module module, uint typeId) + { + UnwrapArray(module, typeId, out uint element, out _, out _); + TypeInfo type = module.Types[element]; + if (type.Op != OpTypeStruct) return false; + for (int index = 0; index < type.Operands.Length; index++) + { + if (module.MemberDecorations.TryGetValue((element, index), out var list) && + list.Exists(d => d.Decoration == DecorationBuiltIn)) + { + return true; + } + } + return false; + } + + /// The GLSL spelling of a non-pointer type; arrays spell only their element. + private static string GlslTypeName(Module module, uint typeId) + { + TypeInfo type = module.Types[typeId]; + switch (type.Op) + { + case OpTypeVoid: + return "void"; + case OpTypeBool: + return "bool"; + case OpTypeInt: + return type.Operands[1] != 0 ? "int" : "uint"; + case OpTypeFloat: + return type.Operands[0] == 64 ? "double" : "float"; + case OpTypeVector: + return ScalarPrefix(module, type.Operands[0]) + "vec" + type.Operands[1]; + case OpTypeMatrix: + { + TypeInfo column = module.Types[type.Operands[0]]; + int rows = (int)column.Operands[1]; + int columns = (int)type.Operands[1]; + string prefix = ScalarPrefix(module, column.Operands[0]) == "d" ? "dmat" : "mat"; + return rows == columns ? prefix + columns : prefix + columns + "x" + rows; + } + case OpTypeSampledImage: + return ImageName(module, type.Operands[0], "sampler"); + case OpTypeImage: + return ImageName(module, typeId, type.Operands[5] == 2 ? "image" : "texture"); + case OpTypeSampler: + return "sampler"; + case OpTypeArray: + case OpTypeRuntimeArray: + return GlslTypeName(module, type.Operands[0]); + case OpTypeStruct: + return NameOf(module, typeId); + default: + return "?"; + } + } + + private static string ScalarPrefix(Module module, uint scalarType) + { + TypeInfo scalar = module.Types[scalarType]; + return scalar.Op switch + { + OpTypeBool => "b", + OpTypeInt => scalar.Operands[1] != 0 ? "i" : "u", + OpTypeFloat when scalar.Operands[0] == 64 => "d", + _ => "", + }; + } + + private static string ImageName(Module module, uint imageTypeId, string stem) + { + uint[] image = module.Types[imageTypeId].Operands; + // sampled type, dim, depth, arrayed, multisampled, sampled, format + string prefix = ScalarPrefix(module, image[0]); + string dim = image[1] switch + { + 0 => "1D", + 1 => "2D", + 2 => "3D", + 3 => "Cube", + 4 => "2DRect", + 5 => "Buffer", + 6 => "SubpassData", + _ => "?", + }; + return prefix + stem + dim + (image[4] != 0 ? "MS" : "") + (image[3] != 0 ? "Array" : "") + + (image[2] == 1 ? "Shadow" : ""); + } + + private static double DecodeScalar(Module module, uint typeId, uint[] literals) + { + TypeInfo type = module.Types[typeId]; + if (literals.Length == 0) return 0; + switch (type.Op) + { + case OpTypeFloat when type.Operands[0] == 32: + return BitConverter.Int32BitsToSingle((int)literals[0]); + case OpTypeFloat: + return BitConverter.Int64BitsToDouble((long)((ulong)literals[0] | (literals.Length > 1 ? (ulong)literals[1] << 32 : 0))); + case OpTypeInt when type.Operands[1] != 0: + return (int)literals[0]; + default: + return literals[0]; + } + } + + private static bool TryConstant(Module module, uint id, out long value) + { + value = 0; + if (!module.Constants.TryGetValue(id, out (uint Type, uint[] Literals) constant) || constant.Literals.Length == 0) + { + return false; + } + TypeInfo type = module.Types[constant.Type]; + value = type.Op == OpTypeInt && type.Operands[1] != 0 ? (int)constant.Literals[0] : constant.Literals[0]; + return true; + } + + private static string NameOf(Module module, uint id) => module.Names.TryGetValue(id, out string? name) ? name : ""; + + private static bool HasDecoration(Module module, uint id, int decoration) => + module.Decorations.TryGetValue(id, out var list) && list.Exists(d => d.Decoration == decoration); + + private static bool TryDecoration(Module module, uint id, int decoration, out uint value) + { + value = 0; + if (!module.Decorations.TryGetValue(id, out var list)) return false; + foreach ((int kind, uint[] literals) in list) + { + if (kind != decoration) continue; + value = literals.Length > 0 ? literals[0] : 0; + return true; + } + return false; + } + + private static bool TryMemberDecoration(Module module, uint type, int member, int decoration, out uint value) + { + value = 0; + if (!module.MemberDecorations.TryGetValue((type, member), out var list)) return false; + foreach ((int kind, uint[] literals) in list) + { + if (kind != decoration) continue; + value = literals.Length > 0 ? literals[0] : 0; + return true; + } + return false; + } + + private static void Add(Dictionary> map, TKey key, (int, uint[]) value) + where TKey : notnull + { + if (!map.TryGetValue(key, out List<(int, uint[])>? list)) + { + list = new List<(int, uint[])>(); + map[key] = list; + } + list.Add(value); + } + + private static string ReadString(uint[] operands, int start, out int next) + { + var bytes = new List(); + int index = start; + for (; index < operands.Length; index++) + { + uint word = operands[index]; + bool terminated = false; + for (int shift = 0; shift < 32; shift += 8) + { + byte value = (byte)(word >> shift); + if (value == 0) + { + terminated = true; + break; + } + bytes.Add(value); + } + if (terminated) break; + } + next = Math.Min(index + 1, operands.Length); + return Encoding.UTF8.GetString(bytes.ToArray()); + } +} diff --git a/Optimum.Render.Vulkan/Transfer/ReadbackManager.cs b/Optimum.Render.Vulkan/Transfer/ReadbackManager.cs new file mode 100644 index 00000000..c7c41429 --- /dev/null +++ b/Optimum.Render.Vulkan/Transfer/ReadbackManager.cs @@ -0,0 +1,194 @@ +using System; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan.Core; + +// The Transfer/ folder follows the plan's layout; the namespace stays Core until +// the renderer is reorganised. + +/// +/// A pending copy of GPU data into a slot's readback arena: valid to read once +/// the Frame timeline passed , and until the slot starts +/// its next frame. +/// +internal readonly record struct ReadbackTicket(VulkanBuffer Buffer, ulong Offset, ulong Size, ulong FrameValue); + +/// +/// Readback inside a frame without ending it. +/// +/// records a barrier and a copy into the current +/// slot's readback arena (a host-visible buffer, bump-allocated, reset when the +/// slot starts a frame). The caller then submits what the frame has recorded +/// with , which continues recording in the +/// same slot with every arena cursor kept, so the frame counter does not move and +/// uniform snapshots stay valid. A caller that needs the bytes now - a +/// screenshot, the parity dump - waits on that one timeline value; nothing else +/// waits. +/// +internal sealed unsafe class ReadbackManager : IDisposable +{ + public const ulong MinimumArenaSize = 1UL << 20; + + // Depth copies need a buffer offset that is a multiple of 4; 8 covers every + // format the dump path reads. + private const ulong OffsetAlignment = 8; + + private readonly VulkanContext _context; + private readonly TextureManager _textures; + private readonly FrameRing _frames; + private readonly VulkanBuffer?[] _arenas; + private readonly ulong[] _cursors; + private bool _disposed; + + public ReadbackManager(VulkanContext context, TextureManager textures, FrameRing frames) + { + _context = context; + _textures = textures; + _frames = frames; + _arenas = new VulkanBuffer?[frames.FramesInFlight]; + _cursors = new ulong[frames.FramesInFlight]; + } + + /// The slot's previous frame has finished and every ticket into its arena was read. + public void BeginSlot(int slotIndex) => _cursors[slotIndex] = 0; + + /// Bytes the slot's arena can hold. Tests only. + internal ulong ArenaCapacity(int slotIndex) => _arenas[slotIndex]?.Size ?? 0; + + /// + /// Records a copy of level 0 of (the given region + /// and aspect) into the current slot's arena, and the transitions around it. + /// The caller has closed any open rendering scope and submits afterwards. + /// + public ReadbackTicket CopyToHost(VulkanTexture texture, int x, int y, uint width, uint height, + ImageAspectFlags aspect, ulong bytes, uint mipLevel = 0) + { + FrameSlot slot = _frames.Current; + CommandBuffer commandBuffer = slot.CommandBuffer; + // The copy writes whole texels whatever the caller asked for, so the + // reservation covers them all; only the requested bytes are handed out. + ulong texel = (ulong)TextureDump.BytesPerTexel(texture.Format); + ulong copied = (ulong)width * height * texel; + VulkanBuffer arena = Reserve(slot.Index, Math.Max(bytes, copied), OffsetAlignmentFor(texel), out ulong offset); + + ImageLayout restore = texture.Layout; + _textures.TransitionTexture(commandBuffer, texture, ImageLayout.TransferSrcOptimal); + + var region = new BufferImageCopy + { + BufferOffset = offset, + ImageSubresource = new ImageSubresourceLayers(aspect, mipLevel, 0, 1), + ImageOffset = new Offset3D(x, y, 0), + ImageExtent = new Extent3D(width, height, 1), + }; + _context.Api.CmdCopyImageToBuffer(commandBuffer, texture.Image, + ImageLayout.TransferSrcOptimal, arena.Handle, 1, ®ion); + + if (restore != ImageLayout.Undefined) _textures.TransitionTexture(commandBuffer, texture, restore); + + return new ReadbackTicket(arena, offset, Math.Min(bytes, copied), slot.FrameValue); + } + + /// + /// A buffer offset legal for a copy of texels of : + /// a multiple of the texel size (VUID-vkCmdCopyImageToBuffer-srcImage-07975, + /// 16 for RGBA32F) and of 4 for depth (-04053). Eight alone put an RGBA32F copy + /// that followed an RGBA8 one at an illegal offset. + /// + internal static ulong OffsetAlignmentFor(ulong texelBytes) + { + ulong alignment = OffsetAlignment; + if (texelBytes == 0) return alignment; + while (alignment % texelBytes != 0) alignment += OffsetAlignment; + return alignment; + } + + /// Whether the copy has run, without waiting. + public bool IsReady(ReadbackTicket ticket) => _frames.Timeline.FrameCompleted >= ticket.FrameValue; + + /// + /// Waits for the ticket's timeline value (counted at the readback site) and + /// copies the bytes out. The ticket's command buffer must have been submitted. + /// + public void WaitAndCopy(ReadbackTicket ticket, IntPtr destination) + { + _frames.Timeline.WaitForFrame(ticket.FrameValue, WaitSite.Readback); + System.Buffer.MemoryCopy((void*)((nint)ticket.Buffer.Mapped + (nint)ticket.Offset), + (void*)destination, (long)ticket.Size, (long)ticket.Size); + } + + private VulkanBuffer Reserve(int slotIndex, ulong bytes, ulong alignment, out ulong offset) + { + VulkanBuffer? arena = _arenas[slotIndex]; + ulong aligned = (_cursors[slotIndex] + alignment - 1) / alignment * alignment; + + if (arena == null || aligned + bytes > arena.Size) + { + ulong size = arena == null ? MinimumArenaSize : arena.Size * 2; + while (size < bytes) size *= 2; + + // A submitted copy may still be writing the old arena and a ticket may + // still name it; the timeline retires it after both. + if (arena != null) _frames.DeferDeletion(arena); + arena = new VulkanBuffer(_context, size, BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, MemoryPoolClass.Staging); + _arenas[slotIndex] = arena; + aligned = 0; + } + + offset = aligned; + _cursors[slotIndex] = aligned + bytes; + return arena; + } + + /// The caller has waited for every signalled frame first. + public void Dispose() + { + if (_disposed) return; + _disposed = true; + for (int i = 0; i < _arenas.Length; i++) + { + _arenas[i]?.Dispose(); + _arenas[i] = null; + } + } +} + +/// +/// The one channel-order conversion the backend owes the client. +/// +/// The client's readback seam is ClientPlatformAbstract.ReadDefaultFramebuffer, +/// and the OpenGL body of it is +/// glReadPixels(..., GL_BGRA, GL_UNSIGNED_BYTE, ...). Its vanilla caller, +/// Screenshot.GrabScreenshot - behind the in-game screenshot key and the AVI +/// recorder - hands that straight to an SKBitmap declared +/// SKColorType.Bgra8888, and Optimum's headless harness writes its PPMs from the +/// same call. The Vulkan device's default colour target is R8G8B8A8_UNORM and its +/// readback copies texels untouched, so without this conversion every Vulkan screenshot, +/// recording and captured frame came out with red and blue exchanged. +/// +/// It lives here, one level above VulkanDevice.ReadDefaultFramebuffer, on +/// purpose: that method is also the backend's general "read the bound target back" +/// operation, which the GPU tests use to inspect attachments in their stored order. +/// Converting there would have changed what every one of those reads means. +/// +internal static class PixelOrder +{ + /// + /// Exchanges the first and third byte of each four-byte texel in place: R G B A + /// becomes B G R A, and back again. Does nothing for a null pointer or a + /// non-positive count. + /// + public static unsafe void SwapRedAndBlue(IntPtr texels, long count) + { + if (texels == IntPtr.Zero || count <= 0L) return; + byte* bytes = (byte*)texels; + for (long i = 0; i < count; i++) + { + byte* texel = bytes + i * 4; + byte first = texel[0]; + texel[0] = texel[2]; + texel[2] = first; + } + } +} diff --git a/Optimum.Render.Vulkan/Transfer/UploadManager.cs b/Optimum.Render.Vulkan/Transfer/UploadManager.cs new file mode 100644 index 00000000..1b9169a1 --- /dev/null +++ b/Optimum.Render.Vulkan/Transfer/UploadManager.cs @@ -0,0 +1,507 @@ +using System; +using System.Collections.Generic; +using System.Threading; +using Silk.NET.Vulkan; + +using Buffer = Silk.NET.Vulkan.Buffer; +using Semaphore = Silk.NET.Vulkan.Semaphore; + +// The Transfer/ folder follows the plan's layout; the namespace stays Core until +// the renderer is reorganised. +namespace Optimum.Render.Vulkan.Core; + +/// Where staged bytes landed: the buffer, the offset inside it, and its host mapping. +internal readonly record struct StagingSlice(Buffer Buffer, ulong Offset, IntPtr Pointer); + +/// +/// Uploads that never wait (transfer backend A of the Vulkan-native plan). +/// +/// Uploads, mip chains and buffer copies are recorded into an open upload batch: +/// a command buffer from the batch's own graphics-family pool plus a bump +/// allocated slice of the staging ring. Any thread may record, under this +/// manager's lock. The next frame submission (a frame end or a partial submit) +/// closes the batch and submits it first, in the same vkQueueSubmit and the same +/// SubmitInfo as the frame's command buffer, so the upload executes before the +/// frame that samples it and nothing waits for it. +/// +/// Each batch reserves a Transfer timeline value when it opens and that +/// submission signals it, alongside the Frame value. Every resource retired +/// while a batch is open is keyed on that Transfer value by the +/// , so a worker's upload can never name a texture or +/// staging buffer that was already destroyed. Every reserved value is signalled +/// by exactly one submission: the frame's, or +/// between frames. +/// +/// GL executes calls in order. The batch runs before the whole frame command +/// buffer, so an upload to a texture or buffer the frame command buffer already +/// used since its last submission would travel back in time past those uses. +/// Those uploads (render thread only; the frame command buffer is the render +/// thread's) are recorded inline into the frame command buffer instead, outside +/// any rendering scope: still no wait, and GL's order holds. +/// +/// Staging: one persistently mapped ring of FramesInFlight x +/// , one region per batch. A batch is reused +/// once the Transfer timeline passed its value. An upload larger than the region's +/// free space, or any upload of a batch beyond the ring's (created when more +/// batches are in flight than the ring has regions), takes a dedicated staging +/// buffer retired on the timelines and counted. +/// +internal sealed unsafe class UploadManager : IDisposable +{ + public const ulong DefaultStagingPerSlot = 32UL << 20; + + // Buffer-to-image copies of depth need offsets that are multiples of 4, and + // wider texels want their own size; 16 covers every format the client uploads. + private const ulong StagingAlignment = 16; + + private sealed class Batch + { + public CommandPool Pool; + public CommandBuffer CommandBuffer; + public ulong RegionStart; + public ulong RegionSize; + public ulong Cursor; + public ulong TransferValue; + public bool Submitted; + public bool EverOpened; + } + + private readonly VulkanContext _context; + private readonly FrameTimeline _timeline; + private readonly RetireQueue _retired; + private readonly int _ringBatches; + private readonly ulong _stagingPerSlot; + private readonly object _lock = new(); + private readonly List _batches = new(); + private VulkanBuffer? _stagingRing; + private Batch? _open; + private bool _disposed; + + // The frame command buffer being recorded, set by the frame slot: the handle + // (0 between submissions), the thread recording it, and a generation that + // changes with every new frame command buffer. Written by the render thread, + // read by any. + private long _frameCommandsHandle; + private int _frameThreadId = -1; + private long _frameGeneration; + + /// + /// Closes the device's open rendering scope on the frame command buffer before + /// an inline upload records transfer commands into it. Null outside a device. + /// + public Action? CloseRenderingScope { get; set; } + + public UploadManager(VulkanContext context, FrameTimeline timeline, RetireQueue retired, + int framesInFlight, ulong stagingPerSlot = DefaultStagingPerSlot) + { + _context = context; + _timeline = timeline; + _retired = retired; + _ringBatches = Math.Max(1, framesInFlight); + _stagingPerSlot = Math.Max(StagingAlignment, stagingPerSlot / StagingAlignment * StagingAlignment); + } + + /// Upload batches created so far (the ring's plus any beyond it). Tests only. + internal int BatchCount + { + get { lock (_lock) return _batches.Count; } + } + + /// Whether a batch is open and will ride the next submission. Tests only. + internal bool HasOpenBatch + { + get { lock (_lock) return _open != null; } + } + + // ------------------------------------------------------------ frame hooks + + /// + /// The frame slot began a new command buffer on the calling thread. Uses + /// noted against the previous one stop counting. + /// + public void OnFrameCommandsStarted(CommandBuffer commandBuffer) + { + Volatile.Write(ref _frameThreadId, Environment.CurrentManagedThreadId); + Interlocked.Increment(ref _frameGeneration); + Volatile.Write(ref _frameCommandsHandle, (long)commandBuffer.Handle); + } + + /// Whether the calling thread is recording a frame command buffer right now. + public bool IsFrameRecordingThread => + Volatile.Read(ref _frameCommandsHandle) != 0 && + Volatile.Read(ref _frameThreadId) == Environment.CurrentManagedThreadId; + + /// Records that , if it is the frame's, uses the texture. + public void NoteUse(CommandBuffer commandBuffer, VulkanTexture texture) + { + if (IsFrameCommands(commandBuffer)) texture.FrameUse = Volatile.Read(ref _frameGeneration); + } + + /// Records that , if it is the frame's, uses the buffer. + public void NoteUse(CommandBuffer commandBuffer, VulkanBuffer buffer) + { + if (IsFrameCommands(commandBuffer)) buffer.FrameUse = Volatile.Read(ref _frameGeneration); + } + + private bool IsFrameCommands(CommandBuffer commandBuffer) => + commandBuffer.Handle != 0 && (long)commandBuffer.Handle == Volatile.Read(ref _frameCommandsHandle); + + /// + /// Whether a resource whose last noted use is is + /// used by the frame command buffer the calling thread is recording, which + /// makes an upload to it an inline one. + /// + public bool UsedByPendingFrame(long frameUse) => + frameUse != 0 && frameUse == Volatile.Read(ref _frameGeneration) && IsFrameRecordingThread; + + // --------------------------------------------------------------- recording + + /// + /// Takes the lock and returns the command buffer to record an upload into: + /// the frame command buffer (scope closed) when + /// holds and the calling thread records the frame, the open batch's otherwise. + /// Pair with in a finally. Reentrant. + /// + public CommandBuffer BeginRecording(bool inlineInFrame) + { + Monitor.Enter(_lock); + try + { + if (inlineInFrame && IsFrameRecordingThread) + { + var frameCommands = new CommandBuffer((nint)Volatile.Read(ref _frameCommandsHandle)); + CloseRenderingScope?.Invoke(frameCommands); + VulkanStats.NoteInlineUpload(); + return frameCommands; + } + return EnsureOpenLocked().CommandBuffer; + } + catch + { + Monitor.Exit(_lock); + throw; + } + } + + public void EndRecording() => Monitor.Exit(_lock); + + /// + /// Bump-allocates staging bytes for the upload being + /// recorded (inside ). Valid until the batch's + /// submission, which carries every command recorded in the meantime, completed. + /// + public StagingSlice Stage(ulong size) + { + if (!Monitor.IsEntered(_lock)) throw new InvalidOperationException("Stage outside BeginRecording"); + + Batch batch = EnsureOpenLocked(); + if (batch.RegionSize > 0) + { + ulong aligned = (batch.Cursor + StagingAlignment - 1) / StagingAlignment * StagingAlignment; + if (aligned + size <= batch.RegionSize) + { + VulkanBuffer ring = StagingRing(); + batch.Cursor = aligned + size; + ulong absolute = batch.RegionStart + aligned; + return new StagingSlice(ring.Handle, absolute, ring.Mapped + (nint)absolute); + } + } + + // Oversized, overflowing, or a batch beyond the ring: its own buffer. The + // open batch's Transfer value (and the newest Frame value, for an inline + // copy) is what the retire entry is keyed on, so it outlives the copy. + var dedicated = new VulkanBuffer(_context, Math.Max(size, 1), BufferUsageFlags.TransferSrcBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, MemoryPoolClass.Staging); + _retired.Retire(dedicated); + VulkanStats.NoteStagingOverflow(); + return new StagingSlice(dedicated.Handle, 0, dedicated.Mapped); + } + + /// + /// Copies bytes into a buffer the host cannot map. Inline when the frame + /// command buffer being recorded already uses the buffer, batched otherwise. + /// + public void UploadToBuffer(VulkanBuffer destination, ulong offset, IntPtr source, ulong size) + { + if (source == IntPtr.Zero || size == 0) return; + + VulkanStats.NoteUploadRequest(); + CommandBuffer commandBuffer = BeginRecording(UsedByPendingFrame(destination.FrameUse)); + try + { + StagingSlice staging = Stage(size); + System.Buffer.MemoryCopy((void*)source, (void*)staging.Pointer, (long)size, (long)size); + + // Buffers have no layouts, so nothing else orders this copy against the + // draws around it: a frame that read the buffer before, and the frame + // command buffer (a later one in the same submission) that reads it + // after. Synchronization validation reports both as hazards without an + // explicit buffer barrier on each side (2026-09-11, the staged index + // buffer of AsyncTransferTests). The other side of each barrier is + // every use the buffer was created for (vertex/index input, uniform or + // storage reads, indirect, transfer), not ALL_COMMANDS. + (PipelineStageFlags2 readerStage, AccessFlags2 readerAccess) = + Graph.BufferUsageState.UsesOf(destination.Usage); + BufferBarrier(commandBuffer, destination, + readerStage, readerAccess, + PipelineStageFlags2.CopyBit, AccessFlags2.TransferWriteBit); + var copy = new BufferCopy { SrcOffset = staging.Offset, DstOffset = offset, Size = size }; + _context.Api.CmdCopyBuffer(commandBuffer, staging.Buffer, destination.Handle, 1, ©); + BufferBarrier(commandBuffer, destination, + PipelineStageFlags2.CopyBit, AccessFlags2.TransferWriteBit, + readerStage, readerAccess); + NoteUse(commandBuffer, destination); + } + finally + { + EndRecording(); + } + } + + /// A synchronization2 barrier on a whole buffer; no queue family change. + private void BufferBarrier(CommandBuffer commandBuffer, VulkanBuffer buffer, + PipelineStageFlags2 sourceStage, AccessFlags2 sourceAccess, + PipelineStageFlags2 destinationStage, AccessFlags2 destinationAccess) + { + var barrier = new BufferMemoryBarrier2 + { + SType = StructureType.BufferMemoryBarrier2, + SrcStageMask = sourceStage, + SrcAccessMask = sourceAccess, + DstStageMask = destinationStage, + DstAccessMask = destinationAccess, + SrcQueueFamilyIndex = Vk.QueueFamilyIgnored, + DstQueueFamilyIndex = Vk.QueueFamilyIgnored, + Buffer = buffer.Handle, + Offset = 0, + Size = Vk.WholeSize, + }; + var dependency = new DependencyInfo + { + SType = StructureType.DependencyInfo, + BufferMemoryBarrierCount = 1, + PBufferMemoryBarriers = &barrier, + }; + _context.Api.CmdPipelineBarrier2(commandBuffer, &dependency); + } + + private VulkanBuffer StagingRing() => + // Allocated on first use: most frame rings in tests never stage anything. + _stagingRing ??= new VulkanBuffer(_context, _stagingPerSlot * (ulong)_ringBatches, + BufferUsageFlags.TransferSrcBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, MemoryPoolClass.Staging); + + private Batch EnsureOpenLocked() + { + if (_disposed) throw new ObjectDisposedException(nameof(UploadManager)); + if (_open != null) return _open; + + Batch? free = null; + ulong completed = 0; + bool completedRead = false; + for (int i = 0; i < _batches.Count; i++) + { + Batch candidate = _batches[i]; + if (!candidate.Submitted) + { + free = candidate; + break; + } + if (!completedRead) + { + completed = _timeline.TransferCompleted; + completedRead = true; + } + if (completed >= candidate.TransferValue) + { + free = candidate; + break; + } + } + + free ??= CreateBatch(); + + Vk api = _context.Api; + if (free.EverOpened) + { + // The Transfer timeline passed the batch's submission: every command + // buffer of the pool has completed, so resetting it is legal. + VulkanResult.Check(api.ResetCommandPool(_context.Device, free.Pool, 0), + "vkResetCommandPool for an upload batch"); + } + + var begin = new CommandBufferBeginInfo + { + SType = StructureType.CommandBufferBeginInfo, + Flags = CommandBufferUsageFlags.OneTimeSubmitBit, + }; + VulkanResult.Check(api.BeginCommandBuffer(free.CommandBuffer, &begin), + "vkBeginCommandBuffer for an upload batch"); + + free.Cursor = 0; + free.Submitted = false; + free.EverOpened = true; + free.TransferValue = _timeline.ReserveTransfer(); + _open = free; + return free; + } + + private Batch CreateBatch() + { + Vk api = _context.Api; + var poolInfo = new CommandPoolCreateInfo + { + SType = StructureType.CommandPoolCreateInfo, + QueueFamilyIndex = _context.GraphicsQueueFamily, + Flags = CommandPoolCreateFlags.TransientBit, + }; + VulkanResult.Check(api.CreateCommandPool(_context.Device, &poolInfo, null, out CommandPool pool), + "vkCreateCommandPool for an upload batch"); + + var allocateInfo = new CommandBufferAllocateInfo + { + SType = StructureType.CommandBufferAllocateInfo, + CommandPool = pool, + Level = CommandBufferLevel.Primary, + CommandBufferCount = 1, + }; + CommandBuffer commandBuffer; + VulkanResult.Check(api.AllocateCommandBuffers(_context.Device, &allocateInfo, &commandBuffer), + "vkAllocateCommandBuffers for an upload batch"); + + int index = _batches.Count; + var batch = new Batch + { + Pool = pool, + CommandBuffer = commandBuffer, + RegionStart = index < _ringBatches ? _stagingPerSlot * (ulong)index : 0, + RegionSize = index < _ringBatches ? _stagingPerSlot : 0, + }; + _batches.Add(batch); + if (index >= _ringBatches) VulkanStats.NoteUploadBatchGrowth(); + return batch; + } + + // -------------------------------------------------------------- submission + + /// + /// Takes the recording lock without recording, for owners whose table changes + /// must be ordered against uploads: a texture deleted under it either was + /// recorded into the open batch first (so its retirement is keyed on that + /// batch's Transfer value) or is not found by an upload that looks it up after. + /// + public void EnterLock() => Monitor.Enter(_lock); + + public void ExitLock() => Monitor.Exit(_lock); + + /// Holds the lock across a frame submission; see . + public void EnterSubmit() => Monitor.Enter(_lock); + + public void ExitSubmit() => Monitor.Exit(_lock); + + /// + /// Inside : closes the open batch, if any, for the + /// submission being built, which must put its command buffer first and signal + /// the Transfer timeline to . + /// + public bool TakeOpenBatchLocked(out CommandBuffer commandBuffer, out ulong transferValue) + { + Batch? batch = _open; + if (batch == null) + { + commandBuffer = default; + transferValue = 0; + return false; + } + + VulkanResult.Check(_context.Api.EndCommandBuffer(batch.CommandBuffer), + "vkEndCommandBuffer for an upload batch"); + batch.Submitted = true; + _open = null; + commandBuffer = batch.CommandBuffer; + transferValue = batch.TransferValue; + return true; + } + + /// + /// Inside , after the queue accepted the submission: + /// the frame command buffer is closed, so nothing can be recorded inline until + /// the slot starts the next one. + /// + public void OnFrameCommandsSubmittedLocked() => Volatile.Write(ref _frameCommandsHandle, 0); + + /// + /// Between frames: submits the open batch on its own (opening an empty one if + /// none is open, so a caller always gets a value to wait on) and returns its + /// Transfer value. Never waits; a readback that needs the bytes waits on the + /// returned value. Refused while a frame is recording: its submission carries + /// the batch, and the batch may hold staging an inline copy reads. + /// + public ulong SubmitStandalone() + { + lock (_lock) + { + if (Volatile.Read(ref _frameCommandsHandle) != 0) + { + throw new InvalidOperationException( + "a frame is being recorded; its submission carries the open upload batch"); + } + + EnsureOpenLocked(); + TakeOpenBatchLocked(out CommandBuffer commandBuffer, out ulong transferValue); + + Semaphore transfer = _timeline.Transfer; + ulong value = transferValue; + var timelineInfo = new TimelineSemaphoreSubmitInfo + { + SType = StructureType.TimelineSemaphoreSubmitInfo, + SignalSemaphoreValueCount = 1, + PSignalSemaphoreValues = &value, + }; + var submit = new SubmitInfo + { + SType = StructureType.SubmitInfo, + PNext = &timelineInfo, + CommandBufferCount = 1, + PCommandBuffers = &commandBuffer, + SignalSemaphoreCount = 1, + PSignalSemaphores = &transfer, + }; + + // Timed like a frame submission: the lock is shared with the swapchain. + long submitStart = VulkanStats.WaitStart(); + lock (_context.QueueLock) + { + VulkanResult.Check(_context.Api.QueueSubmit(_context.GraphicsQueue, 1, &submit, default(Fence)), + "vkQueueSubmit for an upload batch"); + } + VulkanStats.NoteWait(WaitSite.QueueSubmit, submitStart); + _timeline.NoteTransferSubmitted(transferValue); + return transferValue; + } + } + + /// + /// Waits for every signalled Transfer value (teardown only, never throws), + /// then destroys the pools and the staging ring. Retired dedicated staging + /// buffers belong to the retire queue. + /// + public void Dispose() + { + lock (_lock) + { + if (_disposed) return; + _disposed = true; + + _timeline.WaitForSignalledTransfersAtTeardown(); + foreach (Batch batch in _batches) + { + _context.Api.DestroyCommandPool(_context.Device, batch.Pool, null); + } + _batches.Clear(); + _open = null; + _stagingRing?.Dispose(); + _stagingRing = null; + } + } +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.Binding.cs b/Optimum.Render.Vulkan/VulkanDevice.Binding.cs new file mode 100644 index 00000000..e1f12f3e --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.Binding.cs @@ -0,0 +1,699 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan; + +public sealed unsafe partial class VulkanDevice +{ + // ------------------------------------------------------------------- barriers + + /// + /// The frame thread's barriers for sampled textures and feedback snapshots: + /// every texture a draw samples moves in one barrier command. + /// + private Graph.BarrierBatcher _barriers = null!; + + private void SnapshotColorAttachment(CommandBuffer commandBuffer, int textureId, VulkanTexture source) + { + if (_sampledTextureOverrides.ContainsKey(textureId)) return; + + _targets.EndRendering(commandBuffer); + // A pooled ReadSelf copy for this pass (FeedbackCopyPool). + int copyId = _readSelfCopies.Acquire(new Graph.FeedbackCopyDesc(source.Width, source.Height, source.Format, + source.MipLevels, source.Layers, source.Cube)); + VulkanStats.NoteReadSelfCopy(); + VulkanTexture copy = _textures.Get(copyId)!; + copy.State = source.State; + + // Source, copy, and any texture this draw already queued: one command. + _textures.Require(_barriers, commandBuffer, source, Graph.ResourceUsage.TransferSrc); + _textures.Require(_barriers, commandBuffer, copy, Graph.ResourceUsage.TransferDst); + _barriers.Flush(commandBuffer); + for (uint level = 0; level < source.MipLevels; level++) + { + var region = new ImageCopy + { + SrcSubresource = new ImageSubresourceLayers(source.Aspect, level, 0, source.Layers), + DstSubresource = new ImageSubresourceLayers(copy.Aspect, level, 0, copy.Layers), + Extent = new Extent3D(Math.Max(1u, source.Width >> (int)level), + Math.Max(1u, source.Height >> (int)level), 1), + }; + _context.Api.CmdCopyImage(commandBuffer, source.Image, ImageLayout.TransferSrcOptimal, + copy.Image, ImageLayout.TransferDstOptimal, 1, ®ion); + } + _textures.Require(_barriers, commandBuffer, copy, Graph.ResourceUsage.SampleFragment); + _barriers.Flush(commandBuffer); + _sampledTextureOverrides.Add(textureId, copyId); + if (RenderTrace.Enabled) + RenderTrace.Write("snapshot texture=" + textureId + " copy=" + copyId + + " size=" + source.Width + "x" + source.Height + " mips=" + source.MipLevels); + // EnsureRendering transitions the source back to its attachment layout. + } + + /// + /// Copies a client UBO's shadow into this frame's uniform ring, so the draw + /// about to be recorded reads the contents the client uploaded for it rather + /// than whatever the last upload of the frame left behind. + /// + /// One snapshot serves every draw that follows with the block unchanged: the + /// pairing of frame and version is what makes a thousand chunk draws sharing + /// one block cost one copy rather than a thousand. A new frame invalidates it + /// because the ring's cursor is reset. A partial submit does not: the frame + /// stays in the same slot, the cursor keeps counting, and the snapshot's bytes + /// are untouched until that slot starts its next frame. + /// + private bool TrySnapshotClientBlock( + ClientUniformBuffer ubo, ShaderProgramResources program, out uint offset) + { + if (ubo.HasSnapshotFor(_frameCounter)) + { + offset = ubo.SnapshotOffset; + return true; + } + + if (!_frames.Current.TryAllocateUniforms(ubo.Shadow.Length, out RingAllocation allocation)) + { + ReportUniformExhaustion(program, "block '" + ubo.BlockName + "'"); + offset = 0; + return false; + } + + fixed (byte* source = ubo.Shadow) + { + System.Buffer.MemoryCopy(source, (void*)allocation.Pointer, + ubo.Shadow.Length, ubo.Shadow.Length); + } + ubo.NoteSnapshot(_frameCounter, allocation.Offset); + offset = allocation.Offset; + return true; + } + + /// + /// Reports that the frame's uniform ring ran out. Said once per frame so a + /// long frame does not flood the log. + /// + private void ReportUniformExhaustion(ShaderProgramResources program, string what) + { + if (_uniformExhaustionReportedFrame == _frameCounter) return; + + _uniformExhaustionReportedFrame = _frameCounter; + string message = VulkanContext.ErrorPrefix + "uniform ring exhausted in frame " + _frameCounter + + " (" + _frames.Current.UniformBytesUsed + " of " + _frames.Current.UniformCapacity + + " bytes used) at a draw with program " + program.ProgramId + + " '" + ProgramNameOf(program.ProgramId) + "' for " + what; + AddDiagnostic(SanitiseForClientLog(message)); + MirrorValidationMessage(message); + } + + // ----------------------------------------------- shared layout: what is bound + + /// + /// What the current recording holds at each set of the shared pipeline layout (plan + /// decision 9), and the push bytes it last received. Every program's pipelines are + /// built against the one layout, so binding another program's pipeline disturbs none + /// of it: set 1 is bound once per recording, set 0 and set 2 only when their set or + /// dynamic offset changes, and push constants only when the bytes do. A new recording + /// (another serial) or a raw bind outside this path () + /// starts over. + /// + private ulong _boundSerial; + private DescriptorSet _boundFrameSet; + private uint _boundFrameOffset; + private bool _boundTextureSet; + private DescriptorSet _boundStorageSet; + private uint _boundRecordOffset; + private int _pushedLength; + private readonly byte[] _pushShadow = new byte[SetConvention.PushConstantBytes]; + private readonly byte[] _pushedBytes = new byte[SetConvention.PushConstantBytes]; + + /// + /// Set 0's frame textures as the last draw that samples each resolved them, in + /// order; an empty value is that binding's + /// placeholder. Only a program that samples a frame texture reads the binding, and it + /// resolves it again first, so a value left from another program is never read. + /// A texture's deletion clears its values (any thread, hence the lock). + /// + private readonly SamplerBindingValue[] _frameTextureValues = new SamplerBindingValue[SetConvention.FrameTextures.Length]; + + /// The client texture id each entry was resolved from. + private readonly int[] _frameTextureIds = new int[SetConvention.FrameTextures.Length]; + private readonly object _frameTextureLock = new(); + + /// Whether set 1's placeholders have been put in the layout their descriptors name. + private bool _bindlessPlaceholdersReadable; + + /// Binds of set 0 and set 1 this device recorded. Tests only. + internal long FrameSetBindsForTests { get; private set; } + internal long TextureSetBindsForTests { get; private set; } + + /// Forgets what the recording holds bound, after a bind this path did not make. + private void ForgetBoundDescriptors() => _boundSerial = 0; + + private void SyncBoundDescriptors(CommandBuffer commandBuffer) + { + FrameSlot slot = _frames.Current; + ulong serial = slot.CommandBuffer.Handle == commandBuffer.Handle ? slot.RecordingSerial : 0; + if (serial != 0 && serial == _boundSerial) return; + + _boundSerial = serial; + _boundFrameSet = default; + _boundFrameOffset = 0; + _boundTextureSet = false; + _boundStorageSet = default; + _boundRecordOffset = 0; + _pushedLength = 0; + } + + private void ForgetFrameTexture(ulong textureId) + { + lock (_frameTextureLock) + { + for (int i = 0; i < _frameTextureValues.Length; i++) + { + if (_frameTextureValues[i].Resource != textureId) continue; + _frameTextureValues[i] = default; + _frameTextureIds[i] = 0; + } + } + } + + private static int FrameTextureIndex(int binding) + { + for (int i = 0; i < SetConvention.FrameTextures.Length; i++) + { + if (SetConvention.FrameTextures[i].Value == binding) return i; + } + throw new ArgumentOutOfRangeException(nameof(binding), binding, "not a frame texture binding"); + } + + private static TextureKind KindOf(SamplerBinding sampler) + { + if (!sampler.IsFrameTexture) return sampler.Kind; + BindlessKinds.TryFromGlslType(sampler.TypeName, out TextureKind kind); + return kind; + } + + /// Set 0's placeholder for a frame texture: the bindless table's placeholder of the declared kind. + private SamplerBindingValue FrameTexturePlaceholder(int index) + { + SetConvention.Binding frame = SetConvention.FrameTextures[index]; + BindlessKinds.TryFromGlslType(frame.GlslType, out TextureKind kind); + VulkanTexture placeholder = _textures.Get(_bindless!.PlaceholderTextureId(kind))!; + return new SamplerBindingValue((uint)frame.Value, placeholder.View, + _textures.Samplers.Get(BindlessKinds.EffectiveState(SamplerState.Default, kind)), placeholder.Id); + } + + /// + /// Takes a snapshot of the shared frame block when it changed and returns its ring + /// offset. Every draw that follows in the frame reads the same snapshot and only the + /// dynamic offset moves when a value does. + /// + private uint SnapshotFrameGlobals(ShaderProgramResources program) + { + if (_frameGlobalsSnapshotFrame == _frameCounter && _frameGlobalsSnapshotVersion == _frameGlobalsVersion) + { + return _frameGlobalsSnapshotOffset; + } + if (!_frames.Current.TryAllocateUniforms(_frameGlobals.Length, out RingAllocation allocation)) + { + ReportUniformExhaustion(program, "the shared frame block"); + return 0; + } + fixed (byte* source = _frameGlobals) + { + System.Buffer.MemoryCopy(source, (void*)allocation.Pointer, _frameGlobals.Length, _frameGlobals.Length); + } + _frameGlobalsSnapshotFrame = _frameCounter; + _frameGlobalsSnapshotVersion = _frameGlobalsVersion; + _frameGlobalsSnapshotOffset = allocation.Offset; + return allocation.Offset; + } + + /// + /// Binds the three sets of the shared layout for a draw whose push shadow already holds + /// its sampler slots: the frame set, the texture set with the push block, and the storage + /// set. Every native draw binds through here. + /// + private void BindProgramSets(CommandBuffer commandBuffer, ShaderProgramResources program, int meshId) + { + Vk api = _context.Api; + SharedPipelineLayout shared = _sharedLayout!; + SyncBoundDescriptors(commandBuffer); + + // Set 0: the frame block and the fixed frame textures. + if (program.Interface.UsesFrameBlock || program.Interface.UsesFrameTextures) + { + uint offset = SnapshotFrameGlobals(program); + var samplers = new SamplerBindingValue[SetConvention.FrameTextures.Length]; + lock (_frameTextureLock) + { + for (int i = 0; i < samplers.Length; i++) + { + samplers[i] = _frameTextureValues[i].View.Handle != 0 ? _frameTextureValues[i] : FrameTexturePlaceholder(i); + } + } + var contents = new DescriptorSetContents(0, SetConvention.FrameSet, samplers, + new[] + { + new BufferBindingValue((uint)SetConvention.FrameGlobalsBinding, _frames.UniformBuffer, 0, + (ulong)_frameGlobals.Length), + }); + DescriptorSet frameSet = GetDescriptorSet(contents, shared.FrameSetLayout); + if (frameSet.Handle != _boundFrameSet.Handle || offset != _boundFrameOffset) + { + api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Graphics, shared.Layout, + (uint)SetConvention.FrameSet, 1, &frameSet, 1, &offset); + _boundFrameSet = frameSet; + _boundFrameOffset = offset; + FrameSetBindsForTests++; + } + } + + // Set 1 once per recording, and the slot indices when they changed. + int pushSize = program.Interface.PushConstantSize; + if (pushSize > 0) + { + if (!_boundTextureSet) + { + DescriptorSet textureSet = _bindless!.Set; + api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Graphics, shared.Layout, + (uint)SetConvention.TextureSet, 1, &textureSet, 0, null); + _boundTextureSet = true; + TextureSetBindsForTests++; + } + if (pushSize > _pushedLength || + !_pushShadow.AsSpan(0, pushSize).SequenceEqual(_pushedBytes.AsSpan(0, pushSize))) + { + fixed (byte* push = _pushShadow) + { + api.CmdPushConstants(commandBuffer, shared.Layout, SharedPipelineLayout.Stages, 0, (uint)pushSize, push); + } + _pushShadow.AsSpan(0, pushSize).CopyTo(_pushedBytes); + _pushedLength = Math.Max(_pushedLength, pushSize); + VulkanStats.NotePushConstantWrite(); + } + } + + if (program.Interface.UsesStorageSet) + { + BindStorageSet(commandBuffer, program, meshId); + } + } + + /// + /// Set 2, built per draw against the shared layout: the program record at its + /// dynamic binding (a ring snapshot when the shadow changed), each named block from + /// the client's UBO snapshot as a std140 storage buffer at the snapshot's offset, + /// each storage block from the mesh's vertex buffer, and the zero-filled placeholder + /// buffer at every binding the program does not read. A set naming a ring offset is + /// new every frame, so it comes from the slot's arena. + /// + private void BindStorageSet(CommandBuffer commandBuffer, ShaderProgramResources program, int meshId) + { + SharedPipelineLayout shared = _sharedLayout!; + VulkanBuffer placeholder = _placeholderUniforms!; + var buffers = new BufferBindingValue[SetConvention.StorageSetBindingCount]; + for (int binding = 0; binding < buffers.Length; binding++) + { + buffers[binding] = new BufferBindingValue((uint)binding, placeholder.Handle, 0, placeholder.Size, placeholder.Id); + } + + bool namesRingOffset = false; + uint recordOffset = 0; + const int record = SetConvention.ProgramRecordBinding; + + if (program.Interface.HasUniformBlock) + { + if (program.HasSnapshotFor(_frameCounter)) + { + // Nothing written since this program's last draw this frame took its snapshot. + recordOffset = program.SnapshotOffset; + } + else if (_frames.Current.TryAllocateUniforms(program.UniformShadow.Length, out RingAllocation allocation)) + { + fixed (byte* source = program.UniformShadow) + { + System.Buffer.MemoryCopy(source, (void*)allocation.Pointer, + program.UniformShadow.Length, program.UniformShadow.Length); + } + recordOffset = allocation.Offset; + program.NoteSnapshot(_frameCounter, allocation.Offset); + } + else + { + // The draw reads offset zero of the ring, which is some other draw's record: + // wrong, and for a shader that loops on a uniform count, possibly fatal. + ReportUniformExhaustion(program, "its program record"); + } + buffers[record] = new BufferBindingValue(record, _frames.UniformBuffer, 0, (ulong)program.UniformShadow.Length); + } + + // A block the shader declares is fed by whichever UBO the client created under + // that name; one it has not created yet reads the placeholder's zeroes. + foreach (BlockBinding block in program.Interface.UniformBlocks) + { + ClientUniformBuffer? ubo = null; + if (_boundUniformBuffers.TryGetValue(block.BlockName, out int handle)) _uniformBuffers.TryGetValue(handle, out ubo); + if (ubo == null) continue; + + if (TrySnapshotClientBlock(ubo, program, out uint blockOffset)) + { + buffers[block.Binding] = new BufferBindingValue((uint)block.Binding, _frames.UniformBuffer, + blockOffset, (ulong)ubo.Shadow.Length); + namesRingOffset = true; + continue; + } + + // No room left in the ring. Rather than aliasing a buffer every remaining draw + // would share - the exact bug the ring exists to fix - this draw gets its own + // transient copy. Counted, so a scene that lives in this path shows in the stats. + VulkanStats.NoteUniformOverflow(); + var overflow = new VulkanBuffer(_context, (ulong)ubo.Shadow.Length, + BufferUsageFlags.UniformBufferBit | BufferUsageFlags.StorageBufferBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + fixed (byte* shadow = ubo.Shadow) + { + System.Buffer.MemoryCopy(shadow, (void*)overflow.Mapped, ubo.Shadow.Length, ubo.Shadow.Length); + } + buffers[block.Binding] = new BufferBindingValue((uint)block.Binding, overflow.Handle, 0, overflow.Size, overflow.Id); + namesRingOffset = true; + // Released and deferred in that order: no cached set naming it may outlive it. + _descriptors.Release(overflow.Id); + _frames.DeferDeletion(overflow); + } + + // The SSBO chunk path reads its vertices from the mesh's xyz buffer by gl_VertexIndex. + foreach (BlockBinding block in program.Interface.StorageBlocks) + { + VulkanBuffer? buffer = meshId > 0 ? _meshes.BufferOf(meshId, MeshManager.BufferXyz) : null; + if (buffer != null) + { + buffers[block.Binding] = new BufferBindingValue((uint)block.Binding, buffer.Handle, 0, buffer.Size, buffer.Id); + } + else if (RenderTrace.Enabled) + { + RenderTrace.Write("storage block '" + block.BlockName + "' on program " + program.ProgramId + + " has no mesh buffer (mesh " + meshId + "); it reads the placeholder"); + } + } + + if (RenderTrace.Enabled && meshId > 0) + { + // Diagnostic: what this draw binds, so the emulated and the native route can be diffed per draw. + var trace = new System.Text.StringBuilder(" sets program=").Append(program.ProgramId).Append(" mesh=").Append(meshId) + .Append(" record=").Append(recordOffset); + static ulong Fnv(ReadOnlySpan bytes) + { + ulong h = 14695981039346656037UL; + foreach (byte b in bytes) h = (h ^ b) * 1099511628211UL; + return h; + } + foreach (BlockBinding block in program.Interface.UniformBlocks) + { + trace.Append(' ').Append(block.BlockName).Append('@').Append(block.Binding).Append('=') + .Append(buffers[block.Binding].Offset).Append('/').Append(buffers[block.Binding].Resource); + if (_boundUniformBuffers.TryGetValue(block.BlockName, out int traceHandle) && + _uniformBuffers.TryGetValue(traceHandle, out ClientUniformBuffer? traceUbo)) + { + trace.Append(" h").Append(traceHandle).Append(":#").Append(Fnv(traceUbo.Shadow).ToString("x16")); + } + else + { + trace.Append(" (unbound)"); + } + } + trace.Append(" rec#").Append(Fnv(program.UniformShadow).ToString("x16")) + .Append(" push#").Append(Fnv(_pushShadow).ToString("x16")); + trace.Append(" pushBytes=").Append(program.PushShadow?.Length ?? 0) + .Append(" frameBlock=").Append(program.Interface.UsesFrameBlock) + .Append(" frame@").Append(_frameGlobalsSnapshotOffset).Append(" v").Append(_frameGlobalsVersion) + .Append('#').Append(Fnv(_frameGlobals).ToString("x16")); + var inv = System.Globalization.CultureInfo.InvariantCulture; + trace.Append(" recordBytes=").Append(program.UniformShadow.Length); + foreach (UniformMember member in program.Interface.Members) + { + if (member.Name is not ("projectionMatrix" or "viewMatrix" or "modelMatrix")) continue; + if (member.Offset < 0 || member.Offset + 64 > program.UniformShadow.Length) continue; + var f = System.Runtime.InteropServices.MemoryMarshal.Cast(program.UniformShadow.AsSpan(member.Offset, 64)); + trace.Append(' ').Append(member.Name).Append('@').Append(member.Offset).Append("=[") + .Append(f[0].ToString("G4", inv)).Append(',').Append(f[5].ToString("G4", inv)).Append(";t=") + .Append(f[12].ToString("G4", inv)).Append(',').Append(f[13].ToString("G4", inv)).Append(',').Append(f[14].ToString("G4", inv)).Append(']'); + } + RenderTrace.Write(trace.ToString()); + } + + var contents = new DescriptorSetContents(0, SetConvention.StorageSet, Array.Empty(), buffers); + DescriptorSet storageSet = namesRingOffset + ? _descriptorArenas[_frames.Current.Index].Get(contents, shared.StorageSetLayout) + : GetDescriptorSet(contents, shared.StorageSetLayout); + if (storageSet.Handle == _boundStorageSet.Handle && recordOffset == _boundRecordOffset) return; + + _context.Api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Graphics, shared.Layout, + (uint)SetConvention.StorageSet, 1, &storageSet, 1, &recordOffset); + _boundStorageSet = storageSet; + _boundRecordOffset = recordOffset; + VulkanStats.NoteStorageSetBind(); + } + + /// + /// Records the dynamic state a draw needs and the recording does not already hold. The + /// values come from the native pipeline's fixed state and the pass's viewport and scissor. + /// + private void EmitDynamicState(CommandBuffer commandBuffer, DynamicStateValues values, ulong serial, + ColorWriteTier tier, bool dynamicBlend, int colorStates, ReadOnlySpan blendStates) + { + Vk api = _context.Api; + uint colorWrite = values.ColorWrite; + DynamicStateDirty dirty = _dynamicState.Update(serial, values); + if (tier == ColorWriteTier.PipelineKey) dirty &= ~DynamicStateDirty.ColorWrite; + if (!dynamicBlend) dirty &= ~DynamicStateDirty.ColorBlend; + if (dirty == DynamicStateDirty.None) return; + int extraCommands = 0; + + if ((dirty & DynamicStateDirty.ColorWrite) != 0) + { + if (tier == ColorWriteTier.DynamicEnable) + { + // All of maxColorAttachments, so the count covers every pipeline's attachments. + Silk.NET.Core.Bool32* enables = stackalloc Silk.NET.Core.Bool32[colorStates]; + for (int i = 0; i < colorStates; i++) enables[i] = ((colorWrite >> i) & 1) != 0; + _context.ColorWriteEnableApi!.CmdSetColorWriteEnable(commandBuffer, (uint)colorStates, enables); + } + else + { + ColorComponentFlags* masks = stackalloc ColorComponentFlags[colorStates]; + for (int i = 0; i < colorStates; i++) masks[i] = (ColorComponentFlags)((colorWrite >> (i * 4)) & 0xF); + _context.DynamicState3Api!.CmdSetColorWriteMask(commandBuffer, 0, (uint)colorStates, masks); + } + extraCommands++; + } + + if ((dirty & DynamicStateDirty.ColorBlend) != 0) + { + Silk.NET.Core.Bool32* blendEnables = stackalloc Silk.NET.Core.Bool32[colorStates]; + ColorBlendEquationEXT* equations = stackalloc ColorBlendEquationEXT[colorStates]; + for (int i = 0; i < colorStates; i++) + { + AttachmentBlend blend = i < blendStates.Length ? blendStates[i] : AttachmentBlend.Default; + blendEnables[i] = blend.Enabled; + equations[i] = new ColorBlendEquationEXT + { + SrcColorBlendFactor = blend.SrcColor, + DstColorBlendFactor = blend.DstColor, + ColorBlendOp = blend.ColorOp, + SrcAlphaBlendFactor = blend.SrcAlpha, + DstAlphaBlendFactor = blend.DstAlpha, + AlphaBlendOp = blend.AlphaOp, + }; + } + _context.DynamicState3Api!.CmdSetColorBlendEnable(commandBuffer, 0, (uint)colorStates, blendEnables); + _context.DynamicState3Api!.CmdSetColorBlendEquation(commandBuffer, 0, (uint)colorStates, equations); + extraCommands += 2; + } + + if ((dirty & DynamicStateDirty.Viewport) != 0) api.CmdSetViewport(commandBuffer, 0, 1, &values.Viewport); + if ((dirty & DynamicStateDirty.Scissor) != 0) api.CmdSetScissor(commandBuffer, 0, 1, &values.Scissor); + if ((dirty & DynamicStateDirty.CullMode) != 0) api.CmdSetCullMode(commandBuffer, values.CullMode); + if ((dirty & DynamicStateDirty.FrontFace) != 0) api.CmdSetFrontFace(commandBuffer, values.FrontFace); + if ((dirty & DynamicStateDirty.Topology) != 0) api.CmdSetPrimitiveTopology(commandBuffer, values.Topology); + + if ((dirty & DynamicStateDirty.DepthTestEnable) != 0) api.CmdSetDepthTestEnable(commandBuffer, values.DepthTest); + if ((dirty & DynamicStateDirty.DepthWriteEnable) != 0) api.CmdSetDepthWriteEnable(commandBuffer, values.DepthWrite); + if ((dirty & DynamicStateDirty.DepthCompareOp) != 0) api.CmdSetDepthCompareOp(commandBuffer, values.DepthCompare); + + if ((dirty & DynamicStateDirty.StencilTestEnable) != 0) api.CmdSetStencilTestEnable(commandBuffer, values.StencilTest); + if ((dirty & DynamicStateDirty.StencilOp) != 0) + api.CmdSetStencilOp(commandBuffer, StencilFaceFlags.FaceFrontAndBack, + values.StencilFail, values.StencilPass, values.StencilDepthFail, values.StencilCompare); + if ((dirty & DynamicStateDirty.StencilCompareMask) != 0) + api.CmdSetStencilCompareMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, values.StencilCompareMask); + if ((dirty & DynamicStateDirty.StencilWriteMask) != 0) + api.CmdSetStencilWriteMask(commandBuffer, StencilFaceFlags.FaceFrontAndBack, values.StencilWriteMask); + if ((dirty & DynamicStateDirty.StencilReference) != 0) + api.CmdSetStencilReference(commandBuffer, StencilFaceFlags.FaceFrontAndBack, values.StencilReference); + + if ((dirty & DynamicStateDirty.LineWidth) != 0) api.CmdSetLineWidth(commandBuffer, values.LineWidth); + + int emitted = DynamicStateCache.CommandCount(dirty & DynamicStateDirty.All) + extraCommands; + _dynamicStateCommands += emitted; + VulkanStats.NoteDynamicStateCommands(emitted); + } + + /// + /// Commands the first draw of a recording emits: the core set plus the colour + /// write state of the tier (one command; two more with dynamic blend). Tests only. + /// + internal int DynamicStateCommandsPerDrawForTests => + VulkanStats.DynamicStateCommandsPerDraw + + (_context.Capabilities.ColorWriteTier == ColorWriteTier.PipelineKey ? 0 : 1) + + (_context.Capabilities.DynamicColorBlend ? 2 : 0); + + /// The colour write tier this device's draws use. Tests only. + internal ColorWriteTier ColorWriteTierForTests => _context.Capabilities.ColorWriteTier; + + /// vkCmdBeginRendering calls of this device. Tests only. + internal long ScopesOpenedForTests => _targets.ScopesOpened; + + /// A texture's current layout. Tests only. + internal ImageLayout TextureLayoutForTests(int textureId) => + _textures.Get(textureId)?.Layout ?? ImageLayout.Undefined; + + /// Restarts that reopened an identical attachment set; must stay 0. Tests only. + internal long MaskRestartsForTests => _targets.MaskRestarts; + + /// Restarts for a sampled, draw-buffer-excluded slot. Tests only. + internal long FeedbackSplitsForTests => _targets.FeedbackSplits; + + /// Dynamic-state commands this device recorded. Tests only. + internal long DynamicStateCommandsForTests => _dynamicStateCommands; + + /// False emits every dynamic-state command on every draw, as before masking. Tests only. + internal bool DynamicStateMaskingForTests + { + get => _dynamicState.Enabled; + set => _dynamicState.Enabled = value; + } + + /// + /// Routes a set to the current slot's arena when it names a resource created + /// in the last frames (GUI text, + /// atlas tasks, fresh meshes, overflow uniform copies), otherwise to the + /// long-lived cache. + /// + private DescriptorSet GetDescriptorSet(DescriptorSetContents contents, DescriptorSetLayout layout) => + _resourceAge.NamesShortLived(contents) + ? _descriptorArenas[_frames.Current.Index].Get(contents, layout) + : _descriptors.Get(contents, layout); + + /// The frames a resource's sets stay in the arena; 0 sends every set to the cache. Tests only. + internal int ShortLivedFramesForTests + { + get => _resourceAge.ShortLivedFrames; + set => _resourceAge.ShortLivedFrames = value; + } + + /// A slot's descriptor arena. Tests only. + internal DescriptorArena DescriptorArenaForTests(int slot) => _descriptorArenas[slot]; + + /// The slot the current (or last) frame records into. Tests only. + internal int CurrentSlotForTests => _frames.Current.Index; + + /// The indirect ring's bookkeeping. Tests only. + internal IndirectRing IndirectRingForTests => _indirectRing; + + /// Multi-draws that took an overflow buffer, and slot buffers grown at a frame boundary. Tests only. + internal long IndirectOverflowsForTests => _indirectOverflows; + internal long IndirectGrowthsForTests => _indirectGrowths; + + /// Replaces the ring with one whose slot buffers start at bytes. Before the first multi-draw only. Tests only. + internal ulong IndirectMinimumCapacityForTests + { + set => _indirectRing = new IndirectRing(_frames.FramesInFlight, value); + } + + /// + /// The frame boundary of the indirect ring: this slot's cursor returns to 0, + /// and its buffer grows here, and only here, when the busiest frame so far did + /// not fit. Overflow buffers of the frames before retire on the timelines. + /// + private void BeginIndirectFrame(int slot) + { + foreach (VulkanBuffer overflow in _indirectOverflow) _frames.DeferDeletion(overflow); + _indirectOverflow.Clear(); + _indirectOverflowCursor = 0; + + if (_indirectRing.BeginFrame(slot, out ulong capacity)) + { + // Draws of the slot's previous frame named the old buffer; it retires + // on the timelines like any other resource. + _frames.DeferDeletion(_indirectBuffers[slot]!); + _indirectBuffers[slot] = CreateIndirectBuffer(capacity); + _indirectRing.Attach(capacity); + _indirectGrowths++; + } + } + + /// Per-frame dynamic data, so the ReBAR class (a miss falls through, counted). + private VulkanBuffer CreateIndirectBuffer(ulong size) => + new(_context, size, + BufferUsageFlags.IndirectBufferBit, + MemoryPropertyFlags.DeviceLocalBit | MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, + MemoryPoolClass.ReBar); + + /// + /// Hands out a region of the current slot's indirect-command buffer for one + /// multi-draw. + /// + /// The commands are written on the CPU when the draw is recorded and read by + /// the GPU when it executes, which is later - after every other draw of the + /// frame has been recorded too. So each draw needs its own region: writing + /// them all at offset zero meant every multi-draw in a frame executed with + /// the ranges of whichever was recorded last, and the chunk pass is hundreds + /// of them. + /// + /// Regions are bump-allocated per slot and never wrap (Phase 1B step 6): the + /// cursor resets only at the slot's next frame start. A frame that outgrows + /// its slot's buffer continues in an overflow buffer, counted, and the slot + /// grows at its next frame boundary. + /// + private VulkanBuffer AllocateIndirect(int groupCount, out ulong offset) + { + ulong needed = (ulong)Math.Max(groupCount, 1) * (ulong)sizeof(DrawIndexedIndirectCommand); + int slot = _indirectRing.Current; + + if (_indirectRing.NeedsBuffer(needed, out ulong capacity)) + { + // Nothing recorded names a buffer the slot never had, so creating one is safe mid-frame. + _indirectBuffers[slot] = CreateIndirectBuffer(capacity); + _indirectRing.Attach(capacity); + } + + if (_indirectRing.TryAllocate(needed, out offset)) return _indirectBuffers[slot]!; + + _indirectOverflows++; + VulkanStats.NoteIndirectOverflow(); + VulkanBuffer? current = _indirectOverflow.Count == 0 ? null : _indirectOverflow[^1]; + if (current == null || _indirectOverflowCursor + needed > current.Size) + { + current = CreateIndirectBuffer(_indirectRing.CapacityFor(Math.Max(needed, _indirectRing.CapacityOf(slot)))); + _indirectOverflow.Add(current); + _indirectOverflowCursor = 0; + if (RenderTrace.Enabled) + { + RenderTrace.Write("indirect overflow: slot " + slot + " capacity " + _indirectRing.CapacityOf(slot) + + " frame usage " + _indirectRing.FrameUsageOf(slot) + "; overflow buffer " + current.Size); + } + } + + offset = _indirectOverflowCursor; + _indirectOverflowCursor += needed; + return current; + } +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.Native.cs b/Optimum.Render.Vulkan/VulkanDevice.Native.cs new file mode 100644 index 00000000..5d28c314 --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.Native.cs @@ -0,0 +1,1014 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Graph; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; + +namespace Optimum.Render.Vulkan; + +/// Which block of a program's interface a native uniform lives in. +internal enum NativeUniformBlock : byte +{ + None = 0, + /// The program record at set 2, snapshotted into the uniform ring per draw. + Record = 1, + /// A DRAW uniform in the program's push block. + Push = 2, + /// A member of the shared frame block at set 0. + Frame = 3, +} + +/// +/// A uniform's placement in one native pipeline's program, resolved once when the pipeline +/// is created. A native renderer holds the value and writes through it, so no draw looks a +/// uniform up by name. +/// +internal readonly record struct NativeUniform(NativeUniformBlock Block, int Offset, int Size) +{ + public bool IsPresent => Block != NativeUniformBlock.None; +} + +/// +/// A sampler's placement: the push-block offset its bindless slot index is written to (or +/// the set 0 binding, for a fixed frame texture) and the array kind it indexes. Resolved +/// once with the pipeline. +/// +internal readonly record struct NativeSamplerSlot(int Index, int PushOffset, int FrameBinding, TextureKind Kind) +{ + public static NativeSamplerSlot None => new(-1, -1, -1, TextureKind.Texture2D); + + public bool IsPresent => Index >= 0; +} + +/// One sampled texture of a native draw: the slot, the texture handle and the sampler state to read it with (null: the texture's own). +internal readonly record struct NativeTexture(NativeSamplerSlot Sampler, int TextureId, SamplerState? Sampling = null); + +/// +/// The fixed state a native pipeline is built for (docs/vulkan.md, +/// decision 4). It is 's shape stated outright: per-attachment blend and colour write mask, +/// depth test/write/compare, cull, topology and the target's formats. +/// +internal sealed class NativePipelineDescription +{ + /// The linked program: a manifest program when native shaders are on, its rewritten twin otherwise. + public int ProgramId; + + /// The pass name the program was linked under, checked against the device's; null skips the check. + public string? PassName; + + /// The variant the program must have been linked for, checked against the device's; null skips the check. + public string? VariantKey; + + /// Per colour attachment; is the colour write mask. Attachments past the array are not written. + public AttachmentBlend[] Blend = Array.Empty(); + + public bool DepthTest; + public bool DepthWrite; + public CompareOp DepthCompare = CompareOp.Less; + public CullModeFlags Cull = CullModeFlags.None; + + /// + /// The winding a front face has. The game never calls glFrontFace, so every vanilla system + /// states ; a native system that needs the other one + /// says so here rather than through a tracked toggle. + /// + public FrontFace FrontFace = RenderLimits.FrontFace; + + public PrimitiveTopology Topology = PrimitiveTopology.TriangleList; + + /// Fill for every vanilla system; Line is the wireframe debug render's. + public PolygonMode PolygonMode = PolygonMode.Fill; + + /// The width a line-topology draw rasterizes with (autocamera's debug path sets 2). + public float LineWidth = 1.0f; + + /// + /// The vertex layout the pipeline's draws feed it with: + /// for a pass that generates its vertices (the fullscreen triangle), otherwise the layout id of + /// the mesh the system draws (). It is part of the + /// pipeline key, so a mesh pipeline can never be handed a fullscreen one. + /// + public int VertexLayoutId = MeshManager.EmptyLayoutId; + + /// + /// The draws sample the depth attachment of the target they draw into, with depth writes off - + /// what the liquid pass does to fade water at its edges. The scope then holds depth read-only + /// for the draw. Only legal with false; a fullscreen pass leaves it + /// false and sampling its own attachment stays an error. + /// + public bool SamplesBoundDepth; + + /// The attachment formats of the target the pipeline renders into. + public RenderTargetFormats Targets = null!; +} + +/// +/// A pipeline a native render system owns: the program, its fixed state, the pipeline-cache +/// key built from them, and the placement tables the draws write through. +/// +internal sealed class NativePipeline +{ + private readonly Dictionary _uniforms = new(StringComparer.Ordinal); + private readonly Dictionary _samplers = new(StringComparer.Ordinal); + private readonly NativeSamplerSlot[] _samplerSlots; + + internal NativePipeline(ShaderProgramResources program, NativePipelineDescription description, + PipelineKey key, GraphicsPipelineCache.PipelineRequest request, int dynamicBlendId) + { + Program = program; + Description = description; + Key = key; + Request = request; + DynamicBlendId = dynamicBlendId; + + ProgramInterfaceLayout layout = program.Interface; + foreach (UniformMember member in layout.Members) + { + _uniforms[member.Name] = new NativeUniform(NativeUniformBlock.Record, member.Offset, member.Size); + } + foreach (KeyValuePair push in layout.PushMembersByName) + { + _uniforms[push.Key] = new NativeUniform(NativeUniformBlock.Push, push.Value.Offset, push.Value.Size); + } + foreach (string name in layout.FrameMemberDeclaredLengths.Keys) + { + if (FrameGlobals.TryGetMember(name, out UniformMember frame)) + { + _uniforms[name] = new NativeUniform(NativeUniformBlock.Frame, frame.Offset, frame.Size); + } + } + SamplerNames = program.SamplerNames; + _samplerSlots = new NativeSamplerSlot[layout.Samplers.Count]; + for (int i = 0; i < layout.Samplers.Count; i++) + { + SamplerBinding sampler = layout.Samplers[i]; + TextureKind kind = sampler.Kind; + if (sampler.IsFrameTexture) BindlessKinds.TryFromGlslType(sampler.TypeName, out kind); + NativeSamplerSlot slot = new(i, sampler.PushOffset, sampler.FrameBinding, kind); + _samplers[sampler.Name] = slot; + _samplerSlots[i] = slot; + } + } + + /// + /// Every sampler the program declares, in binding order. A system whose draws are all + /// native has to resolve all of them: the emulated resolve that would otherwise fill the + /// push block's slots from the texture units never runs for such a program, so a sampler + /// left out would read whatever slot index was last written there. + /// + internal string[] SamplerNames { get; } + + internal ShaderProgramResources Program { get; } + + public int ProgramId => Program.ProgramId; + + public NativePipelineDescription Description { get; } + + internal PipelineKey Key { get; } + + internal GraphicsPipelineCache.PipelineRequest Request { get; } + + /// The interned blend set the dynamic-state cache compares on, with the mask tier's dynamic blend. + internal int DynamicBlendId { get; } + + /// The placement of a uniform, resolved once here rather than per draw. + public NativeUniform Uniform(string name) => + _uniforms.TryGetValue(name, out NativeUniform member) ? member : default; + + /// The placement of a sampler, resolved once here rather than per draw. + public NativeSamplerSlot Sampler(string name) => + _samplers.TryGetValue(name, out NativeSamplerSlot sampler) ? sampler : NativeSamplerSlot.None; + + /// The declared sampler slot, already resolved when this pipeline was created. + public NativeSamplerSlot SamplerAt(int index) => + (uint)index < (uint)_samplerSlots.Length ? _samplerSlots[index] : NativeSamplerSlot.None; +} + +/// +/// A native pass: an explicit target, the colour slots it writes, the textures it samples +/// and the viewport its draws use. No draw-buffer mask and no bound-target guessing. +/// +internal sealed class NativePassDescription +{ + public string Name = ""; + + /// A render target id, or for the default target. + public int FramebufferId = PassDeclaration.DefaultFramebuffer; + + /// Bit i: colour slot i is an attachment of the pass. + public uint ColorSlots = 1u; + + /// The textures the pass samples, made shader-readable at pass entry. + public int[] Reads = Array.Empty(); + + public uint TransientSlots; + public PassFlags Flags = PassFlags.None; + + /// + /// Bit i: colour slot i starts the pass cleared to . The pass states + /// its own clear instead of a glClearBuffer against a draw-buffer mask, so it lands as the + /// scope's load op. + /// + public uint ClearSlots; + + /// The value clears to. + public float[] ClearValue = { 0f, 0f, 0f, 0f }; + + public int ViewportX; + public int ViewportY; + + /// + /// The scissor the client stated for this draw, in the target's own (GL-oriented) pixels; + /// null: the full target. A GUI draw inside a clipped dialog needs it. + /// + public Rect2D? Scissor; + + /// + /// A pass of the generic stated route (Platform/StatedDraw.cs) rather than of a dedicated + /// native system. It records the same way; only the test counters keep it apart. + /// + public bool Generic; + + /// Negative: the full target. + public int ViewportWidth = -1; + public int ViewportHeight = -1; +} + +/// +/// The device API native render systems draw through (docs/vulkan.md, +/// section 2 decision 4 and section 3). +/// +/// A native system asks for a pipeline by program and fixed state, declares a pass with its +/// target, its colour slots and the textures it reads, writes its uniforms by placement and +/// records draws. Nothing else reaches the GPU: the device has no GL state machine, no +/// texture-unit tables and no draw-buffer mask. Mod renderers and the vanilla systems without a +/// dedicated route draw through the platform's generic native draw (Platform/StatedDraw.cs). +/// +public sealed unsafe partial class VulkanDevice +{ + private readonly Interner _nativeBlends = new(); + private readonly Interner _nativeFormats = new(); + private readonly Dictionary _nativePipelines = new(); + + /// The manifest variant each native program was linked for, keyed by program id. + private readonly Dictionary _programVariants = new(); + + private NativePassDescription? _nativePass; + private VulkanFramebuffer? _nativeTarget; + private long _nativePasses; + private long _nativeDraws; + private long _nativeFullscreenDraws; + private long _nativeMeshDraws; + private long _nativeInstancedDraws; + private long _nativeIndirectDraws; + + /// + /// The identity of a native pipeline: every field of its description that changes what a draw + /// through it does. The vertex layout is in it, so a mesh pipeline never collides with the + /// fullscreen one of the same program, formats and blend; so are the dynamic pieces (front + /// face, line width) that are not in , because the cached + /// carries the description its draws emit. + /// + private readonly record struct NativePipelineCacheKey( + int ProgramId, int FormatsId, int BlendId, bool DepthTest, bool DepthWrite, + CompareOp DepthCompare, CullModeFlags Cull, PrimitiveTopology Topology, + int VertexLayoutId, PolygonMode PolygonMode, FrontFace FrontFace, float LineWidth, + bool SamplesBoundDepth); + + /// Native passes declared and native draws recorded (every kind). Tests only. + internal long NativePassesForTests => _nativePasses; + internal long NativeDrawsForTests => _nativeDraws; + + /// Native draws by kind: the fullscreen triangle, a mesh, an instanced mesh, a multi-draw. Tests only. + internal long NativeFullscreenDrawsForTests => _nativeFullscreenDraws; + internal long NativeMeshDrawsForTests => _nativeMeshDraws; + internal long NativeInstancedDrawsForTests => _nativeInstancedDraws; + internal long NativeIndirectDrawsForTests => _nativeIndirectDraws; + + /// Distinct native pipelines this device holds. Tests only. + internal int NativePipelinesForTests => _nativePipelines.Count; + + /// Set 1's placeholders are written shader-read-only and never used any other way: put them there once. + private void EnsureBindlessPlaceholdersReadable(CommandBuffer commandBuffer) + { + if (_bindlessPlaceholdersReadable || _bindless == null) return; + + for (int kind = 0; kind < BindlessKinds.Count; kind++) + { + VulkanTexture? placeholder = _textures.Get(_bindless.PlaceholderTextureId((TextureKind)kind)); + if (placeholder == null || placeholder.Layout == ImageLayout.ShaderReadOnlyOptimal) continue; + _targets.EndRendering(commandBuffer); + _textures.Require(_barriers, commandBuffer, placeholder, ResourceUsage.SampleFragment); + } + _bindlessPlaceholdersReadable = true; + } + + // ------------------------------------------------------------------ pipelines + + /// + /// The attachment formats of a target's colour slots, so a native system can state the + /// formats its pipeline is built for. Null for a target that does not exist. + /// + internal RenderTargetFormats? NativeTargetFormats(int framebufferId, uint colorSlots) + { + VulkanFramebuffer? target = _targets.Get(ResolveNativeFramebuffer(framebufferId)); + if (target == null) return null; + + return _targets.DeclaredFormats(target, colorSlots); + } + + /// + /// The viewport the GL-shaped state last set. A native pass that keeps the viewport - the + /// OIT merge and sky motion bind their target without touching it, as the OpenGL body's + /// bind-only setter does - states this as its own. + /// + /// + /// Unit assignments in linked sampler order. Writes from SetSamplerUnit and sampler + /// uniform locations update this same array before the next stated draw. + /// + internal int[] NativeSamplerUnitsOf(int programId) + { + return _programs.TryGetValue(programId, out ShaderProgramResources? program) + ? program.SamplerUnitsByIndex + : Array.Empty(); + } + + /// The sampling state of a standalone sampler object (GenSampler), or null. + internal SamplerState? NativeStandaloneSampler(int samplerId) => + _standaloneSamplers.TryGetValue(samplerId, out SamplerState state) ? state : null; + + /// The texture attached at a framebuffer's colour slot, 0 without one. + internal int NativeFramebufferColorTexture(int framebufferId, int slot) + { + VulkanFramebuffer? target = _targets.Get(ResolveNativeFramebuffer(framebufferId)); + return target != null && (uint)slot < (uint)target.Color.Length ? target.Color[slot].TextureId : 0; + } + + /// The depth texture attached to a framebuffer, 0 without one. + internal int NativeFramebufferDepthTexture(int framebufferId) => + _targets.Get(ResolveNativeFramebuffer(framebufferId))?.DepthTextureId ?? 0; + + + /// The manifest variant a program was linked for; "" for a program the rewriter linked. + internal string NativeVariantOf(int programId) => + _programVariants.TryGetValue(programId, out string? key) ? key : ""; + + /// + /// The interned vertex layout of a mesh, which a native system states on the pipeline it + /// draws that mesh through. -1 for a mesh that does not exist. + /// + internal int NativeMeshLayoutId(int meshId) => _meshes.LayoutIdOf(meshId); + + /// + /// The state of a sampler object the client created (glGenSamplers), so a native system + /// can read a program's own sampler override - the chunk terrain's linear sampler on the + /// same atlas texture the nearest sampler reads - straight from the handle the client holds, + /// instead of through the texture unit it was bound to. False for an id that is not one. + /// + internal bool TryNativeSamplerState(int samplerId, out SamplerState state) => + _standaloneSamplers.TryGetValue(samplerId, out state); + + private int ResolveNativeFramebuffer(int framebufferId) => + framebufferId == PassDeclaration.DefaultFramebuffer + ? (_defaultRedirect > 0 ? _defaultRedirect : _defaultFramebuffer) + : framebufferId; + + /// + /// World/UI separation: the target stands for + /// while the platform's UI scope is open - its UI image - or 0 for the window image itself. Every + /// draw, clear, format query and readback that names Default resolves through + /// , so each render system that has always drawn "onto the + /// window" draws into the UI image instead without knowing it, and only the compose, which closes + /// the scope first, writes the window image (VulkanClientPlatform.UiSeparation.cs). + /// + internal void RedirectDefaultFramebuffer(int framebufferId) => _defaultRedirect = framebufferId; + + /// The target Default currently resolves to instead of the window image; 0 for none. + internal int DefaultFramebufferRedirect => _defaultRedirect; + + private int _defaultRedirect; + + /// + /// The pipeline for a program and a piece of fixed state, created through the pipeline + /// cache on first request and returned from this device's table afterwards. Null with a + /// reason when the program is not linked, was linked as something else, or the request + /// names no target formats. + /// + internal NativePipeline? RequestNativePipeline(NativePipelineDescription description, out string error) + { + error = ""; + if (!_programs.TryGetValue(description.ProgramId, out ShaderProgramResources? program)) + { + error = "program " + description.ProgramId + " is not linked"; + return null; + } + if (description.PassName != null && + (!_programNames.TryGetValue(description.ProgramId, out string? name) || + !string.Equals(name, description.PassName, StringComparison.Ordinal))) + { + error = "program " + description.ProgramId + " is not '" + description.PassName + "'"; + return null; + } + if (description.VariantKey != null && + !string.Equals(NativeVariantOf(description.ProgramId), description.VariantKey, StringComparison.Ordinal)) + { + error = "program '" + description.PassName + "' was linked for variant '" + + NativeVariantOf(description.ProgramId) + "', not '" + description.VariantKey + "'"; + return null; + } + if (description.Targets == null) + { + error = "the request names no target formats"; + return null; + } + if (description.SamplesBoundDepth && description.DepthWrite) + { + error = "a pipeline that samples the bound depth attachment cannot also write depth"; + return null; + } + if (description.VertexLayoutId < 0 || description.VertexLayoutId >= _meshes.LayoutCount) + { + error = "vertex layout " + description.VertexLayoutId + " does not exist"; + return null; + } + + ColorWriteTier tier = _context.Capabilities.ColorWriteTier; + bool dynamicBlend = tier == ColorWriteTier.DynamicMask && _context.Capabilities.DynamicColorBlend; + int count = description.Targets.ColorFormats.Length; + + int rawBlendId = _nativeBlends.Intern(new BlendSignature(description.Blend, count)); + int formatsId = _nativeFormats.Intern(description.Targets); + + // Keyed on the blend set as described, not as baked: the draw emits its dynamic blend and + // write masks from the cached description, so two descriptions that bake to one Vulkan + // pipeline under a dynamic tier are still two entries here (they share the PipelineKey below). + var cacheKey = new NativePipelineCacheKey(description.ProgramId, formatsId, rawBlendId, + description.DepthTest, description.DepthWrite, description.DepthCompare, + description.Cull, description.Topology, description.VertexLayoutId, description.PolygonMode, + description.FrontFace, description.LineWidth, description.SamplesBoundDepth); + if (_nativePipelines.TryGetValue(cacheKey, out NativePipeline? cached) && + ReferenceEquals(cached.Program, program)) + { + return cached; + } + + // Baked blend state is owned by a new pipeline request. A cache hit only + // needs the allocation-free raw signature above. + var baked = new AttachmentBlend[Math.Max(count, 1)]; + for (int i = 0; i < baked.Length; i++) + { + AttachmentBlend blend = AttachmentBlend.Default; + if (i < description.Blend.Length) blend = description.Blend[i]; + else blend.WriteMask = 0; + + if (tier == ColorWriteTier.DynamicMask) + { + if (dynamicBlend) blend = AttachmentBlend.Default; + blend.WriteMask = ColorComponentFlags.RBit | ColorComponentFlags.GBit + | ColorComponentFlags.BBit | ColorComponentFlags.ABit; + } + baked[i] = blend; + } + int bakedBlendId = _nativeBlends.Intern(new BlendSignature(baked.AsSpan(0, count))); + + // The system's own vertex layout - the reserved empty one for a pass that generates its + // vertices, the mesh's interned layout for a mesh draw - plus the constant attribute + // defaults GL promises for anything the program declares and the layout does not carry. + VertexLayoutDescription vertexLayout = _meshes.LayoutOf(description.VertexLayoutId) + .WithDefaultsFor(program.Interface.VertexInputs); + + // The blend id is negative, a range of its own in the pipeline cache's key space. The + // vertex layout is in the key, so a mesh pipeline and a fullscreen pipeline of the same + // program are never the same entry. + var key = new PipelineKey( + ProgramId: description.ProgramId, + VertexLayoutId: description.VertexLayoutId, + TargetFormatsId: formatsId, + BlendId: -(bakedBlendId + 1), + PolygonMode: description.PolygonMode, + TopologyClass: GlEnums.TopologyClassOf(description.Topology)); + + var request = new GraphicsPipelineCache.PipelineRequest + { + Program = program, + VertexLayout = vertexLayout, + Targets = description.Targets, + Blend = baked, + PolygonMode = description.PolygonMode, + Topology = description.Topology, + }; + + var pipeline = new NativePipeline(program, description, key, request, -(rawBlendId + 1)); + _nativePipelines[cacheKey] = pipeline; + + // Created here rather than at the first draw where it can be; an async cache queues + // the compile and the first draws are skipped until it is published. + _pipelines.Prepare(key, request); + return pipeline; + } + + /// Whether a pipeline's program is still the linked one of that id (a shader reload replaces it). + internal bool IsNativePipelineLive(NativePipeline pipeline) => + _programs.TryGetValue(pipeline.ProgramId, out ShaderProgramResources? program) && + ReferenceEquals(program, pipeline.Program); + + private void ForgetNativePipelines(int programId) + { + if (_nativePipelines.Count == 0) return; + + var stale = new List(); + foreach (KeyValuePair entry in _nativePipelines) + { + if (entry.Key.ProgramId == programId) stale.Add(entry.Key); + } + foreach (NativePipelineCacheKey key in stale) _nativePipelines.Remove(key); + } + + // ---------------------------------------------------------------------- passes + + /// + /// Opens a native pass on an explicit target: its colour slots, the textures it samples + /// and the viewport its draws use. The draw-buffer mask is not consulted. + /// + internal bool BeginNativePass(NativePassDescription pass) + { + EndNativePass(); + if (!_frameActive) return false; + + int id = ResolveNativeFramebuffer(pass.FramebufferId); + VulkanFramebuffer? target = id > 0 ? _targets.Get(id) : null; + if (target == null) + { + if (RenderTrace.Enabled) RenderTrace.Write("native pass '" + pass.Name + "' skipped: no such target " + id); + return false; + } + + CommandBuffer commandBuffer = Commands; + _targets.DeclarePass(commandBuffer, new PassDeclaration + { + Name = pass.Name, + FramebufferId = id, + ColorSlots = pass.ColorSlots, + Reads = pass.Reads, + TransientSlots = pass.TransientSlots, + Flags = pass.Flags, + }, id); + if (!ReferenceEquals(_targets.Bound, target)) _targets.Bind(commandBuffer, id); + + for (int slot = 0; slot < RenderLimits.MaxColorAttachments && pass.ClearSlots != 0; slot++) + { + if (((pass.ClearSlots >> slot) & 1) == 0) continue; + _targets.ClearPassAttachment(commandBuffer, slot, + pass.ClearValue[0], pass.ClearValue[1], pass.ClearValue[2], pass.ClearValue[3]); + } + + _nativePass = pass; + _nativeTarget = target; + if (!pass.Generic) _nativePasses++; + VulkanStats.NoteNativePass(); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native pass '" + pass.Name + "' target=" + id + " slots=" + pass.ColorSlots + + " reads=" + string.Join(",", pass.Reads)); + } + return true; + } + + /// Closes the native pass and its scope. + internal void EndNativePass() => EndNativePass(keepScope: false); + + /// + /// Closes the native pass. leaves the rendering scope and the + /// pass declaration exactly as they were, for a native draw recorded inside a pass the + /// surrounding stage has already declared - the entity loop, which records one native draw + /// per entity into the Opaque stage's own pass and would otherwise end and restart the + /// rendering scope once per entity. It is only correct when the pass description named that + /// same declaration, so coalesced into it rather than opening + /// one of its own; a pass with its own name, slots or clears must be closed the normal way. + /// + /// The native-pass bookkeeping is cleared either way, so what a render system does between + /// its draws (its uniforms by name) happens outside a native pass. + /// + internal void EndNativePass(bool keepScope) + { + if (_nativePass == null) return; + + _nativePass = null; + _nativeTarget = null; + if (!_frameActive || keepScope) return; + + CommandBuffer commandBuffer = Commands; + _targets.EndPass(commandBuffer); + _targets.EndRendering(commandBuffer); + } + + // --------------------------------------------------------------------- uniforms + + /// Writes a uniform of a native pipeline's program at its resolved placement. + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, ReadOnlySpan data) + { + switch (uniform.Block) + { + case NativeUniformBlock.Record: + pipeline.Program.SetUniform(uniform.Offset, data); + break; + case NativeUniformBlock.Push: + pipeline.Program.SetPushUniform(ShaderProgramResources.PushLocationBase + uniform.Offset, data); + break; + case NativeUniformBlock.Frame: + WriteFrameGlobal(uniform.Offset, data); + break; + } + } + + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, float value) => + WriteNative(pipeline, uniform, new ReadOnlySpan(&value, sizeof(float))); + + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, int value) => + WriteNative(pipeline, uniform, new ReadOnlySpan(&value, sizeof(int))); + + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, float x, float y) + { + float* values = stackalloc float[2] { x, y }; + WriteNative(pipeline, uniform, new ReadOnlySpan(values, 2 * sizeof(float))); + } + + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, float x, float y, float z) + { + float* values = stackalloc float[3] { x, y, z }; + WriteNative(pipeline, uniform, new ReadOnlySpan(values, 3 * sizeof(float))); + } + + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, float x, float y, float z, float w) + { + float* values = stackalloc float[4] { x, y, z, w }; + WriteNative(pipeline, uniform, new ReadOnlySpan(values, 4 * sizeof(float))); + } + + /// + /// A float run at a placement: a matrix, a vector array, a kernel. The model-view matrix a + /// world system writes before each of its draws goes through here, which is what makes the + /// write draw-frequency - a record member is snapshotted into this frame's uniform ring when + /// the draw binds set 2, a push member is pushed with the draw's push block. + /// + internal void WriteNative(NativePipeline pipeline, NativeUniform uniform, ReadOnlySpan values) + { + if (values.IsEmpty) return; + fixed (float* first = values) + { + WriteNative(pipeline, uniform, new ReadOnlySpan(first, values.Length * sizeof(float))); + } + } + + // ------------------------------------------------------------------------ draws + + /// + /// Records the fullscreen triangle of a native pass: the pass's reads are made + /// shader-readable, the sampled textures resolve to bindless slots straight from their + /// handles and sampler state, and the pipeline's fixed state is what the draw runs with. + /// + /// The mesh-drawing siblings are in VulkanDevice.NativeMesh.cs; all of them share + /// , which is this method's old body. + /// + internal bool DrawNativeFullscreen(NativePipeline pipeline, ReadOnlySpan textures) + { + if (!BeginNativeDraw(pipeline, textures, 0, out CommandBuffer commandBuffer, out VulkanFramebuffer? target)) + { + return false; + } + + Checkpoint(commandBuffer, + CheckpointMarker.Draw(CheckpointKind.Fullscreen, pipeline.ProgramId, target!.Id, 0)); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native fullscreen program=" + pipeline.ProgramId + " pass='" + _nativePass!.Name + + "' target=" + target.Id); + } + _context.Api.CmdDraw(commandBuffer, 3, 1, 0, 0); + NoteNativeDraw(NativeDrawKind.Fullscreen); + return true; + } + + /// + /// Everything a native draw needs before its draw command: the pass and pipeline are + /// checked, the textures the draw samples are put into the layout a shader read needs, + /// the rendering scope is opened, the pipeline is bound, the sampled textures resolve to + /// bindless slots straight from their handles and sampler state, the program's sets are + /// bound (with , so a chunk's storage-buffer vertex fetch and an + /// entity's animation block resolve to this draw's mesh) and the dynamic state is emitted. + /// + /// Shared by the fullscreen draw and every mesh draw. None of it reads the GL state + /// tracker, a texture unit or a draw-buffer mask. + /// + private bool BeginNativeDraw(NativePipeline pipeline, ReadOnlySpan textures, int meshId, + out CommandBuffer commandBuffer, out VulkanFramebuffer? target) + { + commandBuffer = default; + target = null; + + if (!_frameActive || _nativePass == null || _nativeTarget == null) + { + if (RenderTrace.Enabled) RenderTrace.Write("native draw skipped: no open native pass"); + return false; + } + + NativePassDescription pass = _nativePass; + VulkanFramebuffer bound = _nativeTarget; + if (!ReferenceEquals(_targets.Bound, bound)) + { + AddDiagnostic("native pass '" + pass.Name + "' lost its target before its draw"); + return false; + } + + ShaderProgramResources program = pipeline.Program; + if (!IsNativePipelineLive(pipeline)) + { + AddDiagnostic("native pass '" + pass.Name + "' draws with program " + pipeline.ProgramId + + ", which has been relinked or deleted"); + return false; + } + + commandBuffer = Commands; + ReleaseReadSelfCopies(); + EnsureBindlessPlaceholdersReadable(commandBuffer); + + // The pass's reads, put into the layout a shader read needs. A colour attachment of + // its own target is sampled through a pooled ReadSelf copy taken before the scope + // opens - the atlas compositions (BlendedTextureManager, RenderTextureIntoFrameBuffer) + // copy one region of an atlas into another region of the same atlas, and the copy is + // what the draw samples (SnapshotColorAttachment). The bound + // depth attachment with depth writes off is sampled in place, which the pipeline + // declares (NativePipelineDescription.SamplesBoundDepth) and the scope then holds + // read-only. + bool depthReadOnly = false; + for (int i = 0; i < textures.Length; i++) + { + VulkanTexture? texture = _textures.Get(textures[i].TextureId); + if (texture == null) continue; + if (_targets.IsBoundDepth(textures[i].TextureId)) + { + if (!pipeline.Description.SamplesBoundDepth) + { + AddDiagnostic("native pass '" + pass.Name + "' samples texture " + textures[i].TextureId + + ", the depth attachment of its own target, through a pipeline that does not declare it"); + return false; + } + depthReadOnly = true; + continue; + } + if (_targets.IsAttachmentOfBound(textures[i].TextureId)) + { + if (texture.Aspect != ImageAspectFlags.ColorBit) + { + AddDiagnostic("native pass '" + pass.Name + "' samples texture " + textures[i].TextureId + + ", a non-colour attachment of its own target"); + return false; + } + SnapshotColorAttachment(commandBuffer, textures[i].TextureId, texture); + continue; + } + _targets.FlushPendingClears(commandBuffer, texture); + if (texture.Layout == ImageLayout.ShaderReadOnlyOptimal) + { + _uploads.NoteUse(commandBuffer, texture); + continue; + } + _targets.EndRendering(commandBuffer); + _textures.Require(_barriers, commandBuffer, texture, ResourceUsage.SampleFragment); + } + PrepareUnnamedFrameTextures(commandBuffer, program, textures); + _barriers.Flush(commandBuffer); + + // Decided before the scope opens, since it decides the depth attachment's layout. + _targets.SetDepthReadOnly(depthReadOnly); + _targets.EnsureRendering(commandBuffer); + + RenderTargetFormats scope = _targets.ScopeFormats(bound); + if (!scope.Equals(pipeline.Description.Targets)) + { + AddDiagnostic("native pass '" + pass.Name + "' has target formats its pipeline was not built for"); + return false; + } + + if (!_pipelines.TryGet(pipeline.Key, pipeline.Request, out Pipeline handle)) + { + if (RenderTrace.Enabled) + { + RenderTrace.Write("native draw skipped: pipeline for program " + pipeline.ProgramId + " still compiling"); + } + return false; + } + + Vk api = _context.Api; + api.CmdBindPipeline(commandBuffer, PipelineBindPoint.Graphics, handle); + + VertexLayoutDescription vertexLayout = pipeline.Request.VertexLayout; + if (vertexLayout.Bindings.Length > 0 && + vertexLayout.Bindings[^1].Binding == VertexLayoutDescription.DefaultAttributeBinding && + _defaultAttributes != null) + { + Silk.NET.Vulkan.Buffer defaults = _defaultAttributes.Handle; + ulong offset = 0; + api.CmdBindVertexBuffers(commandBuffer, + VertexLayoutDescription.DefaultAttributeBinding, 1, &defaults, &offset); + } + + // The program's push block, then this draw's slots over it. + int pushSize = program.Interface.PushConstantSize; + if (program.PushShadow != null) program.PushShadow.CopyTo(_pushShadow, 0); + else if (pushSize > 0) _pushShadow.AsSpan(0, pushSize).Clear(); + + for (int i = 0; i < textures.Length; i++) + { + NativeTexture sampled = textures[i]; + NativeSamplerSlot sampler = sampled.Sampler; + if (!sampler.IsPresent) continue; + + // A ReadSelf copy taken above stands in for the attachment it copies. + VulkanTexture? texture = _textures.Get( + _sampledTextureOverrides.TryGetValue(sampled.TextureId, out int readSelfCopy) + ? readSelfCopy + : sampled.TextureId); + if (texture != null && !BindlessKinds.Suits(TextureShape.Of(texture), sampler.Kind)) + { + if (RenderTrace.Enabled) + { + RenderTrace.Write("native sampler " + sampler.Index + " on program " + program.ProgramId + + " has texture " + sampled.TextureId + " of format " + texture.Format + + " bound, which it cannot sample; using a placeholder"); + } + texture = null; + } + + SamplerState sampling = SamplerState.Default; + if (texture != null) + { + // MAX_LEVEL belongs to the texture, even when the caller overrides the filters. + sampling = sampled.Sampling is { } state + ? state with { MaxLevel = texture.State.MaxLevel } + : texture.State; + } + + // The bound depth attachment is sampled in the read-only depth layout, the one the + // scope holds it in, exactly as the emulated resolve keys it. + ImageLayout layout = depthReadOnly && _targets.IsBoundDepth(sampled.TextureId) + ? ImageLayout.DepthReadOnlyOptimal + : ImageLayout.ShaderReadOnlyOptimal; + + if (sampler.FrameBinding >= 0) + { + SamplerBindingValue value = texture == null + ? default + : new SamplerBindingValue((uint)sampler.FrameBinding, texture.View, + _textures.Samplers.Get(BindlessKinds.EffectiveState(sampling, sampler.Kind)), texture.Id, layout); + if (texture == null) VulkanStats.NoteSamplerPlaceholder(); + int frameIndex = FrameTextureIndex(sampler.FrameBinding); + lock (_frameTextureLock) + { + _frameTextureValues[frameIndex] = value; + _frameTextureIds[frameIndex] = texture == null ? 0 : sampled.TextureId; + } + continue; + } + + uint slot = _bindless!.Resolve(texture, sampler.Kind, sampling, layout); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native sample program=" + program.ProgramId + " sampler=" + sampler.Index + + " logical=" + sampled.TextureId + " physical=" + (texture?.Id ?? 0) + + " slot=" + slot + " layout=" + layout); + } + VulkanStats.NoteBindlessSlotResolution(); + BitConverter.TryWriteBytes(_pushShadow.AsSpan(sampler.PushOffset, ProgramInterfaceLayout.SlotBytes), slot); + } + + BindProgramSets(commandBuffer, program, meshId); + EmitNativeDynamicState(commandBuffer, bound, pass, pipeline); + + target = bound; + return true; + } + + /// + /// A frame texture the program reads (set 0) that this draw does not name keeps the value the + /// last draw that named it left - GL's "whatever the unit still holds". That texture may have + /// been written since (the liquid depth pass renders into liquidDepth's image before the sky + /// dome, whose route names only sky and glow): it is put back into the read layout here, and + /// replaced by the placeholder when it is an attachment of this draw's own target, which a + /// shader cannot read. + /// + private void PrepareUnnamedFrameTextures(CommandBuffer commandBuffer, ShaderProgramResources program, + ReadOnlySpan textures) + { + if (!program.Interface.UsesFrameTextures) return; + foreach (SamplerBinding declared in program.Interface.Samplers) + { + if (!declared.IsFrameTexture) continue; + bool named = false; + for (int i = 0; i < textures.Length && !named; i++) + { + named = textures[i].Sampler.IsPresent && textures[i].Sampler.FrameBinding == declared.FrameBinding; + } + if (named) continue; + + int index = FrameTextureIndex(declared.FrameBinding); + SamplerBindingValue stale; + int textureId; + lock (_frameTextureLock) + { + stale = _frameTextureValues[index]; + textureId = _frameTextureIds[index]; + } + if (stale.View.Handle == 0) continue; + + // Gone, recreated, a ReadSelf copy of an attachment, or an attachment of this draw's own + // target: nothing the shader may read any more, so the placeholder stands in. + VulkanTexture? texture = _textures.Get(textureId); + if (texture == null || texture.Id != stale.Resource || _targets.IsAttachmentOfBound(textureId)) + { + lock (_frameTextureLock) + { + _frameTextureValues[index] = default; + _frameTextureIds[index] = 0; + } + continue; + } + if (stale.Layout != ImageLayout.ShaderReadOnlyOptimal) + { + lock (_frameTextureLock) _frameTextureValues[index] = stale with { Layout = ImageLayout.ShaderReadOnlyOptimal }; + } + _targets.FlushPendingClears(commandBuffer, texture); + if (texture.Layout == ImageLayout.ShaderReadOnlyOptimal) + { + _uploads.NoteUse(commandBuffer, texture); + continue; + } + _targets.EndRendering(commandBuffer); + _textures.Require(_barriers, commandBuffer, texture, ResourceUsage.SampleFragment); + } + } + + /// The dynamic state of a native draw: the pipeline's fixed state and the pass's viewport, never the tracker's. + private void EmitNativeDynamicState(CommandBuffer commandBuffer, VulkanFramebuffer target, + NativePassDescription pass, NativePipeline pipeline) + { + ColorWriteTier tier = _context.Capabilities.ColorWriteTier; + bool dynamicBlend = tier == ColorWriteTier.DynamicMask && _context.Capabilities.DynamicColorBlend; + int colorStates = (int)Math.Min(_context.Capabilities.MaxColorAttachments, (uint)RenderLimits.MaxColorAttachments); + NativePipelineDescription description = pipeline.Description; + + uint colorWrite = 0; + for (int i = 0; i < colorStates && tier != ColorWriteTier.PipelineKey; i++) + { + ColorComponentFlags mask = i < description.Blend.Length ? description.Blend[i].WriteMask : 0; + // An output the program never writes keeps the attachment's contents, as it does on GL. + if (!pipeline.Program.Interface.WrittenFragmentOutputs.Contains(i)) mask = 0; + if (tier == ColorWriteTier.DynamicEnable) + { + if (mask != 0) colorWrite |= 1u << i; + } + else + { + colorWrite |= (uint)mask << (i * 4); + } + } + + int width = pass.ViewportWidth >= 0 ? pass.ViewportWidth : (int)target.Width; + int height = pass.ViewportHeight >= 0 ? pass.ViewportHeight : (int)target.Height; + var values = new DynamicStateValues + { + Viewport = new Viewport(pass.ViewportX, pass.ViewportY, width, height, 0f, 1f), + Scissor = pass.Scissor ?? new Rect2D(new Offset2D(0, 0), new Extent2D(target.Width, target.Height)), + CullMode = description.Cull, + FrontFace = description.FrontFace, + Topology = description.Topology, + DepthTest = description.DepthTest, + DepthWrite = description.DepthWrite, + DepthCompare = description.DepthCompare, + StencilTest = false, + StencilFail = StencilOp.Keep, + StencilPass = StencilOp.Keep, + StencilDepthFail = StencilOp.Keep, + StencilCompare = CompareOp.Always, + StencilCompareMask = 0xFF, + StencilWriteMask = 0xFF, + StencilReference = 0, + // Clamped through the same device range as the emulated path's, so a native line + // draw and the seam's neutral body rasterize identically. + LineWidth = _context.Capabilities.ClampLineWidth(description.LineWidth), + ColorWrite = colorWrite, + BlendStateId = dynamicBlend ? pipeline.DynamicBlendId : 0, + }; + + FrameSlot slot = _frames.Current; + ulong serial = slot.CommandBuffer.Handle == commandBuffer.Handle ? slot.RecordingSerial : 0; + + Span blendStates = stackalloc AttachmentBlend[dynamicBlend ? colorStates : 0]; + for (int i = 0; i < blendStates.Length; i++) + { + blendStates[i] = i < description.Blend.Length ? description.Blend[i] : AttachmentBlend.Default; + } + EmitDynamicState(commandBuffer, values, serial, tier, dynamicBlend, colorStates, blendStates); + } +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.Programs.cs b/Optimum.Render.Vulkan/VulkanDevice.Programs.cs new file mode 100644 index 00000000..744e8565 --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.Programs.cs @@ -0,0 +1,557 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan; + +public sealed unsafe partial class VulkanDevice +{ + // -------------------------------------------------------------------- shaders + + /// + /// Stages a shader. No SPIR-V is produced here because GL resolves uniforms + /// and varyings by name across the whole program, so nothing about a stage is + /// final until its siblings are known. + /// + /// + /// Largest shader source accepted per stage. Vanilla's biggest stage is + /// well under 100 KiB; the cap keeps a broken or hostile mod shader from + /// handing the native compiler an unbounded input. + /// + internal const int MaxShaderSourceBytes = 2 * 1024 * 1024; + + public bool CompileShader(IShader shader) + { + if (shader?.Code == null) return false; + + string stageName = shader.Type.ToString(); + if (shader.Code.Length + (shader.PrefixCode?.Length ?? 0) > MaxShaderSourceBytes) + { + AddDiagnostic($"{stageName}: shader source exceeds {MaxShaderSourceBytes} bytes and was rejected"); + return false; + } + if (shader.Code.IndexOf('\0') >= 0 || (shader.PrefixCode?.IndexOf('\0') ?? -1) >= 0) + { + AddDiagnostic($"{stageName}: shader source contains a NUL byte and was rejected"); + return false; + } + + _stagedStages[shader] = new StagedStage + { + Stage = shader.Type, + Code = shader.Code, + PrefixCode = shader.PrefixCode ?? "", + Filename = shader.Type.ToString(), + }; + return true; + } + + public int LinkProgram(IShaderProgram program) + { + var stages = new List(); + AddStage(stages, program.VertexShader, EnumShaderType.VertexShader, program.PassName); + AddStage(stages, program.FragmentShader, EnumShaderType.FragmentShader, program.PassName); + AddStage(stages, program.GeometryShader, EnumShaderType.GeometryShader, program.PassName); + + if (stages.Count == 0) + { + AddDiagnostic($"shader program '{program.PassName}' has no stages"); + return 0; + } + + // The seam (docs/vulkan.md): a program the manifest has links from its + // SPIR-V; one it has not links through the rewriter; one it has but cannot serve (a bad hash, an + // unreadable define, a module the driver refuses) links through the rewriter and counts as failed. + string passName = program.PassName ?? ""; + ShaderProgramResources? resources = null; + TranslatedProgram? native = null; + bool nativeFailed = false; + string nativeDetail = ""; + if (_nativeShaders != null && _modShaderScan != null && _modShaderScan(passName)) + { + // A mod replaced this program's GLSL (or the scan cannot rule it out): the native SPIR-V would draw + // vanilla over the mod, so the mod's source goes through the rewriter. Rewritten, not failed. + ReportModOverride(passName); + } + else if (_nativeShaders != null) + { + NativeShaderLibrary.Outcome outcome = _nativeShaders.TryLink(passName, stages, out native, out nativeDetail); + nativeFailed = outcome == NativeShaderLibrary.Outcome.Failed; + if (outcome != NativeShaderLibrary.Outcome.Native) native = null; + } + + int programId = 0; + if (native != null) + { + programId = _nextProgramId++; + try + { + resources = new ShaderProgramResources(_context, programId, native, _sharedLayout!.Layout); + } + catch (InvalidOperationException error) + { + nativeFailed = true; + nativeDetail += ": " + error.Message; + } + } + if (nativeFailed) ReportNativeFailure(passName, nativeDetail); + + TranslatedProgram translated; + if (resources != null) + { + translated = native!; + _nativeLinks++; + } + else + { + if (nativeFailed) _failedNativeLinks++; + else _rewrittenLinks++; + + // The include files the registry assembled the program from decide which + // of its uniforms read the shared frame block. A program built any other + // way - a mod's, a test's - keeps every uniform to itself. + translated = ShaderTranslator.Translate(stages, _shaderCompiler, null, + (program as Vintagestory.Client.NoObf.ShaderProgramBase)?.includes); + if (!translated.Success) + { + foreach (string error in translated.Errors) + { + AddDiagnostic($"{program.PassName}: {error}"); + } + return 0; + } + + if (programId == 0) programId = _nextProgramId++; + } + if (RenderTrace.Enabled) + { + RenderTrace.DumpProgramSources(program.PassName, translated); + RenderTrace.Write("program " + programId + " '" + program.PassName + "' uniformBlockBytes=" + + translated.Layout.BlockSize + " pushBytes=" + translated.Layout.PushConstantSize + + (translated.IsNative ? " native [" + nativeDetail + "]" : "")); + foreach (UniformMember member in translated.Layout.Members) + { + RenderTrace.Write(" uniform " + member.Name + " offset=" + member.Offset + + " type=" + member.Type + " count=" + member.ArrayLength); + } + } + resources ??= new ShaderProgramResources(_context, programId, translated, _sharedLayout!.Layout); + _programs[programId] = resources; + // The variant a native program was linked for (TryLink reports the key as its detail), + // so a native pipeline request can state the variant it expects. + if (resources.IsNative) _programVariants[programId] = nativeDetail; + _programNames[programId] = program.PassName ?? ""; + // Pipelines an earlier launch used with this exact program start compiling now. + int prewarming = _pipelines.PrewarmFor(resources); + if (prewarming > 0 && RenderTrace.Enabled) + { + RenderTrace.Write("program " + programId + " prewarming " + prewarming + " pipelines"); + } + return programId; + } + + /// + /// Loads the native shaders once, at device start: the manifest beside the renderer assembly, the directory a + /// test named, or the source tree OPTIMUM_VK_SHADER_SOURCE names. One log line says what came of it. + /// + private void LoadNativeShaders() + { + string? assemblyDirectory = null; + try + { + assemblyDirectory = System.IO.Path.GetDirectoryName(typeof(VulkanDevice).Assembly.Location); + } + catch (Exception error) when (error is ArgumentException or System.IO.PathTooLongException) + { + // An assembly loaded from bytes has no location; the resolution below reports it. + } + + (NativeShaderLibrary.Mode mode, string? path, string reason) = NativeShaderLibrary.Resolve( + NativeShadersEnabled, NativeShaderDirectory, + Environment.GetEnvironmentVariable(NativeShaderLibrary.EnabledVariable), + Environment.GetEnvironmentVariable(NativeShaderLibrary.SourceVariable), + assemblyDirectory); + + _nativeShaders = mode switch + { + NativeShaderLibrary.Mode.Directory => NativeShaderLibrary.Load(path!, _shaderCompiler.Identity, out reason), + NativeShaderLibrary.Mode.Source => NativeShaderLibrary.BuildFromSource(path!, _shaderCompiler, out reason), + _ => null, + }; + + string enabledVariable = Environment.GetEnvironmentVariable(NativeShaderLibrary.EnabledVariable) ?? ""; + bool ignoreScan = IgnoreModShaderScan ?? NativeShaderLibrary.IgnoresModScan(enabledVariable); + _modShaderScan = ignoreScan ? null : ShaderProgramOverriddenByMods; + + NativeShaderStatus = _nativeShaders == null + ? "off: " + reason + : _nativeShaders.Manifest.Programs.Count + " programs from " + _nativeShaders.Origin + + (reason.Length > 0 ? "; " + reason : ""); + if (_nativeShaders != null && ignoreScan && ShaderProgramOverriddenByMods != null) + { + NativeShaderStatus += "; mod shader scan ignored (" + NativeShaderLibrary.EnabledVariable + "=" + NativeShaderLibrary.ForceValue + ")"; + } + else if (_nativeShaders != null && _modShaderScan != null && _modShaderScan(AllShaderProgramsEntry)) + { + NativeShaderStatus += "; the mod shader scan makes every program rewriter-only (no report, a failed scan, or a shaderincludes override)"; + } + LogShaderLine("[Optimum] shaders: native " + NativeShaderStatus); + } + + /// The scan's entry for every program (OptimumConfig.AllShaderPrograms, the scanner's AllPrograms). + internal const string AllShaderProgramsEntry = "all"; + + /// Logs, once per program the manifest has, that the mod shader scan sent it to the rewriter. + private void ReportModOverride(string passName) + { + if (_nativeShaders?.Manifest.FindProgram(passName) == null) return; + bool all = _modShaderScan!(AllShaderProgramsEntry); + lock (_modOverrideLogged) + { + if (!_modOverrideLogged.Add(passName)) return; + } + string line = "[Optimum] shaders: native '" + passName + "' linked through the rewriter: " + + (all ? "the mod shader scan makes every program rewriter-only" : "a mod replaces its GLSL (launcher shader scan)"); + LogShaderLine(line); + if (RenderTrace.Enabled) RenderTrace.Write(line); + } + + private void ReportNativeFailure(string passName, string detail) + { + string line = "[Optimum] shaders: native '" + passName + "' failed, linked through the rewriter: " + detail; + LogShaderLine(line); + if (RenderTrace.Enabled) RenderTrace.Write(line); + } + + /// + /// The line after a shader load: the programs linked since the last report. ShaderRegistry links every program + /// of a load or reload in one synchronous call, so the first frame after links is the end of that load. + /// + private void ReportShaderLoad() + { + int native = _nativeLinks - _nativeLinksReported; + int rewritten = _rewrittenLinks - _rewrittenLinksReported; + int failed = _failedNativeLinks - _failedNativeLinksReported; + if (native + rewritten + failed == 0) return; + + _nativeLinksReported = _nativeLinks; + _rewrittenLinksReported = _rewrittenLinks; + _failedNativeLinksReported = _failedNativeLinks; + LogShaderLine("[Optimum] shaders: " + native + " native, " + rewritten + " rewritten, " + failed + " failed"); + VulkanStats.NoteShaderLoad(native, rewritten, failed); + } + + private static void LogShaderLine(string line) + { + Console.WriteLine(line); + MirrorValidationMessage("--- " + line); + } + + private void AddStage(List stages, IShader? shader, EnumShaderType stage, string passName) + { + if (shader == null || !_stagedStages.TryGetValue(shader, out StagedStage? staged)) return; + + stages.Add(new ShaderStageSource + { + Stage = stage, + Code = staged.Code, + PrefixCode = staged.PrefixCode, + Filename = passName + StageExtension(stage), + }); + } + + private static string StageExtension(EnumShaderType stage) => stage switch + { + EnumShaderType.VertexShader => ".vsh", + EnumShaderType.FragmentShader => ".fsh", + _ => ".gsh", + }; + + public void DeleteProgram(int programId) + { + if (!_programs.Remove(programId, out ShaderProgramResources? program)) return; + _programNames.Remove(programId); + // No background compile may still be reading its modules or layout. + _pipelines.CancelProgram(program); + ForgetNativePipelines(programId); + _frames.DeferDeletion(program); + } + + public int GetUniformLocation(int programId, string name) => + _programs.TryGetValue(programId, out ShaderProgramResources? program) ? program.LocationOf(name) : -1; + + // ------------------------------------------------------------------- uniforms + + private void Write(int programId, int location, ReadOnlySpan data) + { + // A member of the shared frame block: one shadow for every program. + if (ShaderProgramResources.IsFrameLocation(location)) + { + WriteFrameGlobal(location - ShaderProgramResources.FrameLocationBase, data); + return; + } + + if (_programs.TryGetValue(programId, out ShaderProgramResources? program)) + { + // A native program's push member (a DRAW uniform): kept per program, pushed per draw. + if (ShaderProgramResources.IsPushLocation(location)) program.SetPushUniform(location, data); + else program.SetUniform(location, data); + } + } + + /// + /// Writes into the shared frame block. Use() rewrites the same values on every + /// program switch, so the common case is bytes that already match: a comparison + /// and nothing else. A real change bumps the version and the next draw takes a + /// new snapshot. + /// + private void WriteFrameGlobal(int offset, ReadOnlySpan data) + { + if (offset < 0 || offset + data.Length > _frameGlobals.Length) return; + + Span destination = _frameGlobals.AsSpan(offset, data.Length); + if (data.SequenceEqual(destination)) return; + + data.CopyTo(destination); + _frameGlobalsVersion++; + } + + /// A copy of a linked program's record shadow, initializers included; null for an unknown program. For tests. + internal byte[]? ProgramRecordForTests(int programId) => + _programs.TryGetValue(programId, out ShaderProgramResources? program) ? (byte[])program.UniformShadow.Clone() : null; + + /// A copy of the shared frame block's current bytes. For tests. + + public void SetUniform(int programId, int location, float value) => + Write(programId, location, new ReadOnlySpan(&value, sizeof(float))); + + public void SetUniform(int programId, int location, int value) + { + // Assigning a sampler its texture unit is an int write to its uniform + // location in GL. Here the sampler is a descriptor binding, so the same + // call has to reach the unit table instead of the uniform block. + if (ShaderProgramResources.IsSamplerLocation(location)) + { + if (_programs.TryGetValue(programId, out ShaderProgramResources? program)) + { + program.SetSamplerUnitByLocation(location, value); + } + return; + } + Write(programId, location, new ReadOnlySpan(&value, sizeof(int))); + } + + public void SetUniform(int programId, int location, int x, int y, int z) + { + // Scalar block layout stores an ivec3 as three consecutive 32-bit ints. + int* values = stackalloc int[3] { x, y, z }; + Write(programId, location, new ReadOnlySpan(values, 3 * sizeof(int))); + } + + public void SetUniform(int programId, int location, float x, float y) + { + float* values = stackalloc float[2] { x, y }; + Write(programId, location, new ReadOnlySpan(values, 2 * sizeof(float))); + } + + public void SetUniform(int programId, int location, float x, float y, float z) + { + float* values = stackalloc float[3] { x, y, z }; + Write(programId, location, new ReadOnlySpan(values, 3 * sizeof(float))); + } + + public void SetUniform(int programId, int location, float x, float y, float z, float w) + { + float* values = stackalloc float[4] { x, y, z, w }; + Write(programId, location, new ReadOnlySpan(values, 4 * sizeof(float))); + } + + // The array setters are a straight memcpy because the generated block uses + // scalar layout, where a float[] packs exactly as the shader expects. Under + // std140 each of these would need re-striding on the way in. + private void WriteArray(int programId, int location, int count, float[] values, int componentsPerElement) + { + int floats = Math.Min(values.Length, count * componentsPerElement); + if (floats <= 0) return; + + fixed (float* source = values) + { + Write(programId, location, new ReadOnlySpan(source, floats * sizeof(float))); + } + } + + public void SetUniformArray1(int programId, int location, int count, float[] values) => + WriteArray(programId, location, count, values, 1); + + public void SetUniformArray2(int programId, int location, int count, float[] values) => + WriteArray(programId, location, count, values, 2); + + public void SetUniformArray3(int programId, int location, int count, float[] values) => + WriteArray(programId, location, count, values, 3); + + public void SetUniformArray4(int programId, int location, int count, float[] values) => + WriteArray(programId, location, count, values, 4); + + public void SetUniformMatrix(int programId, int location, float[] matrix) => + WriteArray(programId, location, 1, matrix, 16); + + public void SetUniformMatrices(int programId, int location, int count, float[] matrices) => + WriteArray(programId, location, count, matrices, 16); + + public void SetUniformMatrices4x3(int programId, int location, int count, float[] matrices) => + WriteArray(programId, location, count, matrices, 12); + + /// + /// The samplers a linked program declares, in declaration order. + /// + /// Not part of the seam - the client never needs it, because it binds the + /// samplers it knows by name. It exists so a test can bind every sampler a + /// real program declares without hardcoding the list, since a draw whose + /// descriptor set is incomplete is skipped rather than drawn. + /// + internal string[] SamplerNamesOf(int programId) + { + if (_programs.TryGetValue(programId, out ShaderProgramResources? program)) + { + return program.SamplerNames; + } + return Array.Empty(); + } + + public void SetSamplerUnit(int programId, string samplerName, int unit) + { + if (_programs.TryGetValue(programId, out ShaderProgramResources? program)) + { + program.SetSamplerUnitByName(samplerName, unit); + } + } + + // ------------------------------------------------------------ uniform buffers + + /// + /// One uniform buffer object the client created for a named block. + /// + /// The CPU shadow is the source of truth, not the GPU buffer. The client + /// updates one UBO per block and re-updates it between draws - the entity + /// renderer uploads the "Animation" block once per entity, immediately before + /// that entity's draw - but a draw is only recorded here, not executed, so a + /// buffer written in place would give every entity in the frame the last + /// entity's transforms. Writes therefore land in ordinary memory and a draw + /// snapshots them into the frame's uniform ring, exactly as the generated + /// block does. + /// + /// There is deliberately no GPU buffer per block: the snapshot goes in the + /// ring, and the ring-exhausted path allocates its own transient copy for + /// that one draw, so a persistent buffer would only ever sit unbound. + /// + private sealed class ClientUniformBuffer + { + public ClientUniformBuffer(byte[] shadow, string blockName) + { + Shadow = shadow; + BlockName = blockName; + } + + public byte[] Shadow { get; } + public string BlockName { get; } + + /// Bumped by every write, so an unchanged block reuses its snapshot. + public uint Version { get; private set; } = 1; + + /// Which frame's ring the snapshot below lives in, and what it holds. + public uint SnapshotFrame { get; private set; } + public uint SnapshotVersion { get; private set; } + public uint SnapshotOffset { get; private set; } + + public void Write(IntPtr data, int offset, int size) + { + // A client that re-uploads identical bytes before every draw would + // otherwise cost a fresh ring slice per draw; comparing is cheaper. + var incoming = new ReadOnlySpan((void*)data, size); + Span target = Shadow.AsSpan(offset, size); + if (incoming.SequenceEqual(target)) return; + incoming.CopyTo(target); + Version++; + } + + public void NoteSnapshot(uint frame, uint offset) + { + SnapshotFrame = frame; + SnapshotVersion = Version; + SnapshotOffset = offset; + } + + public bool HasSnapshotFor(uint frame) => SnapshotFrame == frame && SnapshotVersion == Version; + } + + private readonly Dictionary _uniformBuffers = new(); + + /// + /// The buffer currently supplying each named block. + /// + /// The client's UBO binds with glBindBufferBase to binding point 0 and names + /// the block when it creates the buffer, so the block name is what actually + /// identifies which declaration a buffer feeds. Vulkan has no such global + /// binding point, so the association is kept here and resolved per draw + /// against the program's own declared blocks. + /// + private readonly Dictionary _boundUniformBuffers = new(StringComparer.Ordinal); + + private int _nextUniformBufferId = 1; + + public int CreateUniformBuffer(int programId, int bindingPoint, string blockName, int size) + { + int bytes = Math.Max(size, 4); + int id = _nextUniformBufferId++; + _uniformBuffers[id] = new ClientUniformBuffer(new byte[bytes], blockName ?? ""); + + // GL's glBindBufferBase in the client's constructor takes effect at once, + // and a buffer is only ever created to be used. + if (!string.IsNullOrEmpty(blockName)) _boundUniformBuffers[blockName] = id; + return id; + } + + public void UpdateUniformBuffer(int handle, IntPtr data, int offset, int size) + { + if (!_uniformBuffers.TryGetValue(handle, out ClientUniformBuffer? ubo)) return; + if (data == IntPtr.Zero || offset < 0 || size < 0) return; + if ((long)offset + size > ubo.Shadow.Length) return; + + ubo.Write(data, offset, size); + } + + public void BindUniformBuffer(int handle) + { + if (_uniformBuffers.TryGetValue(handle, out ClientUniformBuffer? ubo) && ubo.BlockName.Length > 0) + { + _boundUniformBuffers[ubo.BlockName] = handle; + } + } + + /// + /// Deliberately does not break the block association. + /// + /// The client's Unbind is glBindBuffer(UNIFORM_BUFFER, 0), which clears the + /// generic target and leaves the glBindBufferBase index binding standing - + /// and the index binding is what feeds the shader. Dropping the association + /// here would unbind the block the client still expects to be supplied. + /// + public void UnbindUniformBuffer(int handle) { } + + public void DeleteUniformBuffer(int handle) + { + if (!_uniformBuffers.Remove(handle, out ClientUniformBuffer? ubo)) return; + + if (ubo.BlockName.Length > 0 && + _boundUniformBuffers.TryGetValue(ubo.BlockName, out int bound) && bound == handle) + { + _boundUniformBuffers.Remove(ubo.BlockName); + } + } +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.Readback.cs b/Optimum.Render.Vulkan/VulkanDevice.Readback.cs new file mode 100644 index 00000000..2eea7622 --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.Readback.cs @@ -0,0 +1,279 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan; + +public sealed unsafe partial class VulkanDevice +{ + // -------------------------------------------------------------------- queries + + private QueryRing _queryRing = null!; + private ReadbackManager _readbacks = null!; + + /// Whether occlusion queries count samples exactly. Tests only. + internal bool PreciseOcclusionForTests => _context.Capabilities.OcclusionQueryPrecise; + + /// Occlusion query pools across every frame slot. Tests only. + internal int OcclusionQueryPoolsForTests => _queryRing.PoolCount; + + public int CreateOcclusionQuery() => _queryRing.Create(); + + public void BeginOcclusionQuery(int queryId) + { + if (!_frameActive || !_queryRing.CanBegin(queryId)) return; + + // The slot's pools are reset at frame start, before any scope opens, so + // the query begins inside the scope the covered draw uses. Only a pool + // created just now needs a reset here, and a reset has to happen outside + // a scope: one restart per pool ever, never in steady state. If the + // scope later closes before the query ends, the ring's scope hooks + // suspend it and resume it in the next scope. + CommandBuffer commandBuffer = Commands; + if (_queryRing.NextNeedsPool) + { + _targets.EndRendering(commandBuffer); + _queryRing.AddPool(commandBuffer); + } + // No scope is opened for it: a query begun outside one is suspended and starts in the next + // scope that opens - the native pass of the draw it covers (QueryRing.OnScopeOpened). + _queryRing.Begin(queryId, commandBuffer, _targets.RenderingActive); + } + + public void EndOcclusionQuery(int queryId) + { + if (_frameActive) _queryRing.End(queryId, _frames.Current.FrameValue, Commands); + } + + /// + /// GL_QUERY_RESULT_AVAILABLE without any wait: true once the Frame timeline + /// passed the command buffer that ended the query, a frame or two later. + /// + public bool IsQueryResultAvailable(int queryId) => _queryRing.IsResultAvailable(queryId); + + /// + /// The samples the latest query counted. Never waits and never submits: the + /// client polls availability first (sun glare does), and a result asked for + /// early returns the previous query's count, or "all visible" if there was + /// none - for a query that gates culling or glare, the cheap failure. + /// + public int GetQueryResult(int queryId) => _queryRing.GetResult(queryId); + + /// + /// Submits everything the frame has recorded so far and keeps recording it in + /// the same slot, so a readback queued next sees work the frame already issued. + /// The open upload batch rides along, first. No wait, no new slot, no frame + /// counter increment: arena cursors and uniform snapshots carry on. + /// + private ulong SubmitPartial() + { + _targets.EndRendering(Commands); + _bindless?.Flush(); + ulong submitted = _frames.SubmitPartial(); + Checkpoint(Commands, CheckpointMarker.FrameBegin(_frameCounter)); + return submitted; + } + + public void DeleteQuery(int queryId) => _queryRing.Delete(queryId); + + // ------------------------------------------------------------------- readback + + /// + /// Reads back every texture OPTIMUM_DUMP_TEXTURES asked for. Debug only; see + /// for why it exists. + /// + private void DumpRequestedTextures() + { + foreach (int textureId in TextureDump.Take()) + { + VulkanTexture? texture = _textures.Get(textureId); + if (texture == null) + { + RenderTrace.Write("texture dump: no texture " + textureId); + continue; + } + + int width = (int)texture.Width; + int height = (int)texture.Height; + byte[] data = ReadBackLevel0(texture); + + bool bgra = texture.Format is Format.B8G8R8A8Unorm or Format.B8G8R8A8Srgb; + bool written = TextureDump.Write(textureId, width, height, bgra, texture.Format, data); + if (written) TextureDump.Complete(textureId); + + RenderTrace.Write("texture dump: " + textureId + " " + width + "x" + height + + " " + texture.Format + " mips=" + texture.MipLevels + " -> " + (written ? "ok" : "failed")); + } + } + + /// + /// Copies level 0 of a texture into host memory, raw texels in the image's own + /// format, rows in memory order (GL order: the backend never flips Y). Inside + /// a frame it goes through , so the frame stays open. + /// Depth images are copied through their depth aspect. + /// + /// Level 0 of a texture through the dump path's readback. Tests only. + internal byte[] ReadBackLevel0ForTests(int textureId) => + ReadBackLevel0(_textures.Get(textureId) ?? throw new ArgumentException("no texture " + textureId)); + + /// One mip level of a texture through the in-frame readback. Tests only; a frame must be open. + internal byte[] ReadBackLevelForTests(int textureId, uint mipLevel) + { + VulkanTexture texture = _textures.Get(textureId) ?? throw new ArgumentException("no texture " + textureId); + if (!_frameActive) throw new InvalidOperationException("a mip readback needs an open frame"); + uint width = Math.Max(1, texture.Width >> (int)mipLevel); + uint height = Math.Max(1, texture.Height >> (int)mipLevel); + ulong bytes = (ulong)width * height * (ulong)BytesPerPixel(texture.Format); + var data = new byte[bytes]; + _targets.FlushPendingClears(Commands, texture); + _targets.EndRendering(Commands); + ReadbackTicket ticket = _readbacks.CopyToHost(texture, 0, 0, width, height, texture.Aspect, bytes, mipLevel); + SubmitPartial(); + fixed (byte* destination = data) _readbacks.WaitAndCopy(ticket, (IntPtr)destination); + return data; + } + + private byte[] ReadBackLevel0(VulkanTexture texture) + { + int width = (int)texture.Width; + int height = (int)texture.Height; + ulong bytes = (ulong)width * (ulong)height * (ulong)BytesPerPixel(texture.Format); + ImageAspectFlags aspect = (texture.Aspect & ImageAspectFlags.DepthBit) != 0 + ? ImageAspectFlags.DepthBit + : ImageAspectFlags.ColorBit; + + byte[] data = new byte[bytes]; + fixed (byte* destination = data) + { + ReadBack(texture, 0, 0, (uint)width, (uint)height, aspect, bytes, (IntPtr)destination); + } + return data; + } + + /// + /// The one readback path: screenshots, the texture dump and the parity dump. + /// + /// Inside a frame the open scope closes, the copy is recorded into the frame + /// itself (into the slot's readback arena), the recorded part is submitted + /// with and the caller waits on that single Frame + /// timeline value; the frame carries on in the same slot, so every draw after + /// the read still reaches the screen. Between frames the copy is appended to + /// the open upload batch (after every upload recorded so far), which is + /// submitted on its own; the queue runs it after every frame already submitted, + /// so waiting on its Transfer value is enough. Neither path waits for the + /// whole device. + /// + private void ReadBack(VulkanTexture texture, int x, int y, uint width, uint height, + ImageAspectFlags aspect, ulong bytes, IntPtr destination) + { + if (_frameActive) + { + _targets.FlushPendingClears(Commands, texture); + _targets.EndRendering(Commands); + ReadbackTicket ticket = _readbacks.CopyToHost(texture, x, y, width, height, aspect, bytes); + SubmitPartial(); + _readbacks.WaitAndCopy(ticket, destination); + return; + } + + // The copy writes whole texels of the image's format whatever the caller + // sized its destination for; the buffer holds them all, the caller gets its bytes. + ulong copied = (ulong)width * height * (ulong)BytesPerPixel(texture.Format); + ulong handed = Math.Min(bytes, copied); + using var readback = new VulkanBuffer(_context, Math.Max(bytes, copied), + BufferUsageFlags.TransferDstBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit, MemoryPoolClass.Staging); + + ImageLayout restore = texture.Layout; + CommandBuffer commandBuffer = _uploads.BeginRecording(inlineInFrame: false); + try + { + _textures.TransitionTexture(commandBuffer, texture, ImageLayout.TransferSrcOptimal); + + var region = new BufferImageCopy + { + ImageSubresource = new ImageSubresourceLayers(aspect, 0, 0, 1), + ImageOffset = new Offset3D(x, y, 0), + ImageExtent = new Extent3D(width, height, 1), + }; + _context.Api.CmdCopyImageToBuffer(commandBuffer, texture.Image, + ImageLayout.TransferSrcOptimal, readback.Handle, 1, ®ion); + + if (restore != ImageLayout.Undefined) _textures.TransitionTexture(commandBuffer, texture, restore); + } + finally + { + _uploads.EndRecording(); + } + ulong transferValue = _uploads.SubmitStandalone(); + _frames.Timeline.WaitForTransfer(transferValue, WaitSite.Readback); + + System.Buffer.MemoryCopy((void*)readback.Mapped, (void*)destination, (long)bytes, (long)handed); + } + + private void RecordGlInternalFormat(int textureId, int glInternalFormat) + { + VulkanTexture? texture = _textures.Get(textureId); + if (texture != null) texture.GlInternalFormat = glInternalFormat; + } + + /// + /// The parity dump's readback (): level 0 in + /// the representation glGetTexImage produces on the OpenGL path, decoded by + /// . Debug only. + /// + public OptimumTextureReadback? ReadTextureForParity(int textureId) + { + if (!_frameActive) return null; + VulkanTexture? texture = _textures.Get(textureId); + if (texture == null || texture.Cube || texture.Layers > 1) return null; + + byte[] data = ReadBackLevel0(texture); + int glInternalFormat = texture.GlInternalFormat != 0 + ? texture.GlInternalFormat + : TextureDump.GlInternalFormatOf(texture.Format); + OptimumTextureReadback? readback = TextureDump.ToParityReadback(texture.Format, glInternalFormat, + (int)texture.Width, (int)texture.Height, data); + RenderTrace.Write("parity dump: texture " + textureId + " " + texture.Width + "x" + texture.Height + + " " + texture.Format + " -> " + (readback != null ? "ok" : "undecodable")); + return readback; + } + + /// + /// Bytes per texel for the formats the dump path is expected to see. + /// Shared with 's decode switch so the + /// readback size and the reader always agree on the stride. + /// + private static int BytesPerPixel(Format format) => TextureDump.BytesPerTexel(format); + + /// Colour attachment 0 of an explicit target (the default one for ). + internal void ReadFramebufferColor(int framebufferId, int x, int y, int width, int height, IntPtr destination) + { + EndNativePass(); + ReadFramebufferColor(_targets.Get(ResolveNativeFramebuffer(framebufferId)), x, y, width, height, destination); + } + + private void ReadFramebufferColor(VulkanFramebuffer? target, int x, int y, int width, int height, IntPtr destination) + { + if (destination == IntPtr.Zero || width <= 0 || height <= 0) return; + if (target == null) return; + + VulkanTexture? texture = _textures.Get(target.Color[0].TextureId); + if (texture == null) return; + + ReadBack(texture, x, y, (uint)width, (uint)height, ImageAspectFlags.ColorBit, + (ulong)width * (ulong)height * 4, destination); + } + + /// + /// The format of the default colour target, so the platform above knows the + /// channel order the readback hands back rather than assuming one. + /// + internal Format DefaultColorFormat => DefaultColorTexture()?.Format ?? Format.R8G8B8A8Unorm; +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.Resources.cs b/Optimum.Render.Vulkan/VulkanDevice.Resources.cs new file mode 100644 index 00000000..aeb346e4 --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.Resources.cs @@ -0,0 +1,796 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan; + +public sealed unsafe partial class VulkanDevice +{ + // -------------------------------------------------------------------- textures + + public int CreateTexture2D( + int width, int height, EnumTextureInternalFormat internalFormat, + EnumTexturePixelFormat pixelFormat, IntPtr pixels, bool generateMipmaps) + { + int id = _textures.Create((uint)width, (uint)height, + GlEnums.TextureFormatFrom(internalFormat), generateMipmaps: generateMipmaps); + RecordGlInternalFormat(id, (int)internalFormat); + + if (pixels != IntPtr.Zero) + { + _textures.Upload(id, 0, 0, 0, (uint)width, (uint)height, pixels, BytesPerPixel(internalFormat)); + if (generateMipmaps) _textures.GenerateMipmaps(id); + } + return id; + } + + public int CreateTexture2DRaw(int width, int height, int glInternalFormat, IntPtr pixels, int bytesPerPixel, + bool generateMipmaps = false) + { + Format format = GlEnums.TextureFormatFromGl(glInternalFormat); + int id = _textures.Create((uint)width, (uint)height, format, generateMipmaps: generateMipmaps); + RecordGlInternalFormat(id, glInternalFormat); + + if (pixels != IntPtr.Zero && bytesPerPixel > 0) + { + _textures.Upload(id, 0, 0, 0, (uint)width, (uint)height, pixels, bytesPerPixel); + + // A chain that was asked for has to be filled here. GL's texture is + // complete the moment glGenerateMipmap runs, but an image created + // with levels and never blitted into keeps whatever its memory held, + // and every sample above level 0 reads that - which looks like other + // textures bleeding onto a surface as it turns away from the camera. + if (generateMipmaps) _textures.GenerateMipmaps(id); + } + RenderTrace.TextureCreated(id, width, height, format, pixels, bytesPerPixel); + return id; + } + + /// + /// A post-chain colour texture (framebuffer slots in + /// ): created in the Transient + /// memory pool class and registered with the transient allocator. Until the frame + /// graph binds it () it behaves like any texture. + /// + public int CreateTransientTexture2D(int width, int height, EnumTextureInternalFormat internalFormat, + int framebufferSlot) + { + int id = _textures.Create((uint)width, (uint)height, GlEnums.TextureFormatFrom(internalFormat), + poolClass: MemoryPoolClass.Transient); + RecordGlInternalFormat(id, (int)internalFormat); + _transients.OptIn(id, framebufferSlot); + return id; + } + + /// Registers (or re-tags) a texture as the transient of client framebuffer slot . + public void OptInTransient(int textureId, int framebufferSlot) => _transients.OptIn(textureId, framebufferSlot); + + /// with a raw GL internal format token, no pixels. + public int CreateTransientTexture2DRaw(int width, int height, int glInternalFormat, int framebufferSlot) + { + Format format = GlEnums.TextureFormatFromGl(glInternalFormat); + int id = _textures.Create((uint)width, (uint)height, format, poolClass: MemoryPoolClass.Transient); + RecordGlInternalFormat(id, glInternalFormat); + RenderTrace.TextureCreated(id, width, height, format, IntPtr.Zero, 0); + _transients.OptIn(id, framebufferSlot); + return id; + } + + /// + /// Serves a texture for passes [, ] + /// of the current frame through the transient allocator and returns the texture id + /// that backs it (itself unless aliasing is on). Call after BeginFrame, in pass order. + /// + public int BindTransientForFrame(int textureId, int firstPass, int lastPass) => + _transients.Bind(textureId, firstPass, lastPass); + + /// The transient allocator the frame graph acquires physical images from. + internal Graph.TransientAllocator Transients => _transients; + + /// The ReadSelf copy pool. Tests only. + internal Graph.FeedbackCopyPool ReadSelfCopiesForTests => _readSelfCopies; + + /// Forces transient aliasing on or off before Initialize (default: OPTIMUM_VULKAN_ALIAS). + internal bool? TransientAliasingOverride { get; set; } + + private int CreateReadSelfCopy(Graph.FeedbackCopyDesc desc) => + _textures.Create(desc.Width, desc.Height, desc.Format, layers: desc.Layers, cube: desc.Cube, + generateMipmaps: desc.MipLevels > 1, poolClass: MemoryPoolClass.Transient); + + /// Gives the previous draw's ReadSelf copies back to the pool. + private void ReleaseReadSelfCopies() + { + if (_sampledTextureOverrides.Count == 0) return; + foreach (int copy in _sampledTextureOverrides.Values) _readSelfCopies.Release(copy); + _sampledTextureOverrides.Clear(); + } + + public int CreateTextureCubeRaw(int size, int glInternalFormat, IntPtr[] facePixels, int bytesPerPixel) + { + Format format = GlEnums.TextureFormatFromGl(glInternalFormat); + int id = _textures.Create((uint)size, (uint)size, format, cube: true); + + for (uint face = 0; face < 6 && face < facePixels.Length; face++) + { + if (facePixels[face] == IntPtr.Zero) continue; + _textures.Upload(id, 0, 0, 0, (uint)size, (uint)size, + facePixels[face], bytesPerPixel, face); + } + return id; + } + + public int CreateTextureCube( + int size, EnumTextureInternalFormat internalFormat, + EnumTexturePixelFormat pixelFormat, IntPtr[] facePixels) + { + Format format = GlEnums.TextureFormatFrom(internalFormat); + int id = _textures.Create((uint)size, (uint)size, format, cube: true); + + for (uint face = 0; face < 6 && face < facePixels.Length; face++) + { + if (facePixels[face] == IntPtr.Zero) continue; + _textures.Upload(id, 0, 0, 0, (uint)size, (uint)size, + facePixels[face], BytesPerPixel(internalFormat), face); + } + return id; + } + + public int CreateTexture2DArray( + int width, int height, int layers, + EnumTextureInternalFormat internalFormat, EnumTexturePixelFormat pixelFormat) => + _textures.Create((uint)width, (uint)height, + GlEnums.TextureFormatFrom(internalFormat), layers: (uint)layers); + + public void UploadTexture2D( + int textureId, int level, int x, int y, int width, int height, + EnumTexturePixelFormat pixelFormat, IntPtr pixels) + { + FlushPendingClears(textureId); + _textures.Upload(textureId, level, x, y, (uint)width, (uint)height, pixels, + pixelFormat == EnumTexturePixelFormat.Red ? 1 : 4); + } + + public void UploadTexture2DRaw( + int textureId, int level, int x, int y, int width, int height, IntPtr pixels, int bytesPerPixel) + { + if (bytesPerPixel <= 0) return; + FlushPendingClears(textureId); + _textures.Upload(textureId, level, x, y, (uint)width, (uint)height, pixels, bytesPerPixel); + } + + public void GenerateMipmaps(int textureId) + { + FlushPendingClears(textureId); + _textures.GenerateMipmaps(textureId); + } + + public void DeleteTexture(int textureId) => ReleaseTexture(textureId); + + /// + /// Deletes a texture and evicts every descriptor set that names it. + /// + /// The eviction is the important half. The texture itself is destroyed once + /// the Frame timeline passed every frame that could name it, but a cached set would outlive it and, once the driver + /// reused the view handle for a new texture, be served to draws of that new + /// texture - which is a GPU read of freed memory. The GUI re-renders its text + /// into fresh textures constantly, so this was the loading-screen crash. + /// + private void ReleaseTexture(int textureId) + { + if (_sampledTextureOverrides.Remove(textureId, out int copy)) _readSelfCopies?.Release(copy); + _transients?.Forget(textureId); + _textures.RestoreBinding(textureId); + VulkanTexture? texture = _textures.Get(textureId); + if (texture != null) + { + _descriptors.Release(texture.Id); + _targets.DropPendingClears(texture); + } + _textures.Delete(textureId, _frames); + // Only a delete that found something is a delete. Deleting an id twice + // (framebuffers share a depth texture) otherwise inflated the counter + // past the number of textures that ever existed. + if (texture != null) VulkanStats.NoteTextureDeleted(); + } + + public void SetTextureParameter(int textureId, int parameterName, int value) => + _textures.SetParameter(textureId, parameterName, value); + + public void SetTextureParameter(int textureId, int parameterName, float value) => + _textures.SetParameter(textureId, parameterName, value); + + public void SetTextureBorderColor(int textureId, float r, float g, float b, float a) => + _textures.SetBorderColor(textureId, r, g, b, a); + + public int GetTextureParameter(int textureId, int parameterName) + { + VulkanTexture? texture = _textures.Get(textureId); + if (texture == null) return 0; + + return parameterName == GlEnums.TextureCompareMode + ? texture.State.CompareEnable ? GlEnums.TextureCompareRefToTexture : GlEnums.TextureCompareModeNone + : 0; + } + + public void UploadTexture2DArrayLayer(int textureId, int layer, int x, int y, + int width, int height, IntPtr pixels) + { + FlushPendingClears(textureId); + _textures.Upload(textureId, 0, x, y, (uint)width, (uint)height, pixels, 4, (uint)layer); + } + + public void UploadTexture2DNormalizedShorts(int textureId, int level, int x, int y, + int width, int height, short[] pixels) + { + FlushPendingClears(textureId); + _textures.UploadNormalizedShorts(textureId, level, x, y, width, height, pixels); + } + + private readonly Dictionary _standaloneSamplers = new(); + private int _nextSamplerId = 1; + + public int CreateSampler(bool linear) + { + int id = _nextSamplerId++; + _standaloneSamplers[id] = SamplerState.Default with + { + MagFilter = linear ? Filter.Linear : Filter.Nearest, + // GenSampler uses GL_NEAREST_MIPMAP_LINEAR for both variants; + // the flag changes magnification only. Terrain relies on this + // override retaining the atlas mip chain at a distance. + MinFilter = Filter.Nearest, + MipmapMode = SamplerMipmapMode.Linear, + Mipmapped = true, + }; + return id; + } + + public void SetSamplerParameter(int samplerId, int parameterName, float value) + { + if (!_standaloneSamplers.TryGetValue(samplerId, out SamplerState state)) return; + + _standaloneSamplers[samplerId] = parameterName == GlEnums.TextureLodBias + ? state with { LodBias = value } + : state; + } + + public void DeleteSampler(int samplerId) => _standaloneSamplers.Remove(samplerId); + + private static int BytesPerPixel(EnumTextureInternalFormat format) => format switch + { + EnumTextureInternalFormat.Rgba8 => 4, + EnumTextureInternalFormat.Rgba16f => 8, + EnumTextureInternalFormat.R16f => 2, + EnumTextureInternalFormat.DepthComponent32 => 4, + _ => 4, + }; + + // ---------------------------------------------------------------- framebuffers + + public int CreateFramebuffer(int width, int height) => _targets.Create((uint)width, (uint)height); + + public void AttachTexture(int framebufferId, EnumFramebufferAttachment attachment, int textureId, int layer) + { + int index = attachment == EnumFramebufferAttachment.DepthAttachment + ? -1 + : (int)attachment - (int)EnumFramebufferAttachment.ColorAttachment0; + + _targets.Attach(framebufferId, index, textureId, (uint)layer); + } + + public bool CheckFramebufferComplete(int framebufferId, out string status) + { + // Dynamic rendering has no framebuffer object to validate, so + // completeness reduces to having a target with attachments. + VulkanFramebuffer? framebuffer = _targets.Get(framebufferId); + if (framebuffer == null) + { + status = "no such framebuffer"; + return false; + } + + status = "complete"; + return true; + } + + public void DeleteFramebuffer(int framebufferId) + { + _targets.Delete(framebufferId); + FramebufferDeleted?.Invoke(framebufferId); + } + + /// The platform whose graphics this device is; null for a bare device (the GPU tests). + internal Platform.VulkanClientPlatform? OwnerPlatform { get; set; } + + /// + /// Raised after a framebuffer is deleted. Its id is reused by the next one created, so a + /// record keyed on it (the stated draw buffers) has to forget it here. + /// + internal Action? FramebufferDeleted; + + /// Whether the frame graph records this device's frames. Change only between frames. + internal bool FrameGraphEnabled + { + get => _graph.Enabled; + set => _graph.Enabled = value; + } + + /// The frame graph's totals. Tests only. + internal Graph.FrameGraph FrameGraphForTests => _graph; + + /// The bound render target's id (0 before any bind). + internal int BoundFramebufferId => _targets.Bound?.Id ?? 0; + + /// The render target standing for the default framebuffer (0 when headless). + internal int DefaultFramebufferId => _defaultFramebuffer; + + /// + /// Ends the pass a render stage left open (closes its scope): a native draw recorded with + /// keepScope stays in the stage's declaration until here. No-op with the frame graph off. + /// + internal void EndStagePass() + { + if (_frameActive) _targets.EndPass(Commands); + } + + /// Lands the clears promoted into a texture before it is written some other way. + private void FlushPendingClears(int textureId) + { + if (!_frameActive || !_graph.HasPendingClears) return; + VulkanTexture? texture = _textures.Get(textureId); + if (texture != null) _targets.FlushPendingClears(Commands, texture); + } + + /// + /// A colour clear of one attachment of an explicit target, outside every native pass: the + /// promoted LOAD_OP_CLEAR of the next pass on it, or an attachment clear inside an open scope. + /// The caller has applied the draw buffers and colour mask it stated (VulkanClientPlatform). + /// + internal void ClearNativeColor(int framebufferId, int attachment, float r, float g, float b, float a) + { + if (!BindForNativeClear(framebufferId)) return; + if (RenderTrace.Enabled) + { + RenderTrace.Write("clearColor attachment=" + attachment + " target=" + _targets.Bound!.Id + + " rgba=" + r + "," + g + "," + b + "," + a); + } + _targets.ClearColor(Commands, attachment, r, g, b, a); + } + + /// The depth clear of an explicit target; the caller has applied the stated depth mask. + internal void ClearNativeDepth(int framebufferId, float depth) + { + if (!BindForNativeClear(framebufferId)) return; + if (RenderTrace.Enabled) RenderTrace.Write("clearDepth target=" + _targets.Bound!.Id + " depth=" + depth); + _targets.ClearDepth(Commands, depth); + } + + private bool BindForNativeClear(int framebufferId) + { + EndNativePass(); + if (!_frameActive) return false; + int id = ResolveNativeFramebuffer(framebufferId); + VulkanFramebuffer? target = _targets.Get(id); + if (target == null) return false; + if (!ReferenceEquals(_targets.Bound, target)) _targets.Bind(Commands, id); + return true; + } +} + +public sealed unsafe partial class VulkanDevice +{ + // --------------------------------------------------------------------- meshes + + public int CreateMesh(MeshData data, bool staticDraw) + { + // Sized as GL's UploadMesh sizes them. Every part follows the vertex + // count except flags, which GL allocates at the array's full length - + // a mesh that later grows within that capacity updates its flags in + // place there, and would overflow a vertex-count-sized buffer here. + int vertices = data.VerticesCount; + int id = _meshes.CreateEmpty( + data.xyz != null ? vertices * 3 * sizeof(float) : 0, + data.Normals != null ? vertices * sizeof(int) : 0, + data.Uv != null ? vertices * 2 * sizeof(float) : 0, + data.Rgba != null ? vertices * 4 : 0, + data.Flags != null ? data.Flags.Length * sizeof(int) : 0, + data.IndicesCount * sizeof(int), + data.CustomFloats, data.CustomShorts, data.CustomBytes, data.CustomInts, + data.mode, staticDraw, ssbo: false, signedCustomShorts: true); + + UpdateMesh(id, data); + return id; + } + + public int CreateEmptyMesh( + int xyzSize, int normalsSize, int uvSize, int rgbaSize, int flagsSize, int indicesSize, + CustomMeshDataPartFloat customFloats, CustomMeshDataPartShort customShorts, + CustomMeshDataPartByte customBytes, CustomMeshDataPartInt customInts, + EnumDrawMode drawMode, bool staticDraw, bool ssbo) => + _meshes.CreateEmpty(xyzSize, normalsSize, uvSize, rgbaSize, flagsSize, indicesSize, + customFloats, customShorts, customBytes, customInts, drawMode, staticDraw, ssbo); + + /// + /// Writes a mesh's data, honouring the destination offset each part carries. + /// + /// Those offsets are the whole point. The game pools chunk meshes: one large + /// mesh holds many chunks, and each chunk is handed the same mesh with the + /// byte offset of its own slice in every part. GL's updateVAO takes that + /// offset as the destination for a glBufferSubData, so writing it at zero + /// instead stacks every chunk in the world on top of the first one - which + /// renders as no terrain at all. + /// + /// The counts are per part as well, not VerticesCount: a part can be absent + /// or shorter than the vertex count, and the custom buffers have no fixed + /// relationship to it. + /// + public void UpdateMesh(int meshId, MeshData data) + { + // An SSBO mesh's xyz slot holds packed face records, written through + // UpdateMeshStorageBuffer; positions never belong there. The game hands + // the same MeshData to both calls, so without this the positions would + // land on top of the records - or, depending on order, under them. + bool ssbo = _meshes.IsSsbo(meshId); + + if (data.xyz != null && data.XyzCount > 0 && !ssbo) + { + fixed (float* source = data.xyz) + { + _meshes.Write(meshId, MeshManager.BufferXyz, data.XyzOffset, + (IntPtr)source, data.XyzCount * sizeof(float)); + } + } + // The normals, uv and flags streams have no buffer on an SSBO mesh - the + // face records carry what the shader needs from them - so GL's SSBO + // update path never writes them either. + if (data.Normals != null && data.VerticesCount > 0 && !ssbo) + { + fixed (int* source = data.Normals) + { + _meshes.Write(meshId, MeshManager.BufferNormals, data.NormalsOffset, + (IntPtr)source, data.VerticesCount * sizeof(int)); + } + } + if (data.Uv != null && data.UvCount > 0 && !ssbo) + { + fixed (float* source = data.Uv) + { + _meshes.Write(meshId, MeshManager.BufferUv, data.UvOffset, + (IntPtr)source, data.UvCount * sizeof(float)); + } + } + if (data.Rgba != null && data.RgbaCount > 0) + { + fixed (byte* source = data.Rgba) + { + _meshes.Write(meshId, MeshManager.BufferRgba, data.RgbaOffset, + (IntPtr)source, data.RgbaCount); + } + } + if (data.Flags != null && data.FlagsCount > 0 && !ssbo) + { + fixed (int* source = data.Flags) + { + _meshes.Write(meshId, MeshManager.BufferFlags, data.FlagsOffset, + (IntPtr)source, data.FlagsCount * sizeof(int)); + } + } + if (data.CustomFloats != null && data.CustomFloats.Count > 0) + { + fixed (float* source = data.CustomFloats.Values) + { + _meshes.Write(meshId, MeshManager.BufferCustomFloat, data.CustomFloats.BaseOffset, + (IntPtr)source, data.CustomFloats.Count * sizeof(float)); + } + } + if (data.CustomShorts != null && data.CustomShorts.Count > 0) + { + fixed (short* source = data.CustomShorts.Values) + { + _meshes.Write(meshId, MeshManager.BufferCustomShort, data.CustomShorts.BaseOffset, + (IntPtr)source, data.CustomShorts.Count * sizeof(short)); + } + } + if (data.CustomInts != null && data.CustomInts.Count > 0) + { + if (ssbo) + { + WritePrunedCustomInts(meshId, data.CustomInts); + } + else + { + fixed (int* source = data.CustomInts.Values) + { + _meshes.Write(meshId, MeshManager.BufferCustomInt, data.CustomInts.BaseOffset, + (IntPtr)source, data.CustomInts.Count * sizeof(int)); + } + } + } + if (data.CustomBytes != null && data.CustomBytes.Count > 0) + { + fixed (byte* source = data.CustomBytes.Values) + { + _meshes.Write(meshId, MeshManager.BufferCustomByte, data.CustomBytes.BaseOffset, + (IntPtr)source, data.CustomBytes.Count); + } + } + // An SSBO mesh never takes indices from the data: GL draws every such + // mesh through one shared index buffer holding the fixed quad pattern, + // filled once at allocation, and its update path leaves indices alone. + // The mesh here got the same pattern when it was created. + if (data.Indices != null && data.IndicesCount > 0 && !ssbo) + { + fixed (int* source = data.Indices) + { + _meshes.Write(meshId, -1, data.IndicesOffset, + (IntPtr)source, data.IndicesCount * sizeof(int)); + } + } + } + + /// + /// Writes the custom ints as the SSBO path stores them: two per vertex go in + /// and only the second of each pair is kept, the first being the colormap + /// data that the face record already carries. The destination offset halves + /// with the stride. A part with a single int per vertex is not bound at all + /// on this path, so there is nothing to write. + /// + private void WritePrunedCustomInts(int meshId, CustomMeshDataPartInt customInts) + { + if (customInts.InterleaveStride <= 4) return; + + int kept = customInts.Count / 2; + if (kept <= 0) return; + + if (_prunedCustomInts.Length < kept) _prunedCustomInts = new int[kept]; + + int[] values = customInts.Values; + for (int i = 0; i < kept; i++) _prunedCustomInts[i] = values[i * 2 + 1]; + + fixed (int* source = _prunedCustomInts) + { + _meshes.Write(meshId, MeshManager.BufferCustomInt, customInts.BaseOffset / 2, + (IntPtr)source, kept * sizeof(int)); + } + } + + /// + /// The SSBO chunk path packs four vertices into one face record and stores + /// them in the xyz slot, which CreateEmptyMesh gave StorageBufferBit usage + /// and no vertex-attribute binding when ssbo was set. Writing it is a plain + /// buffer write; the shader reads it through gl_VertexIndex. + /// + public void UpdateMeshStorageBuffer(int meshId, IntPtr data, int byteOffset, int byteSize) => + _meshes.Write(meshId, MeshManager.BufferXyz, byteOffset, data, byteSize); + + public IntPtr GetMappedPointer(int meshId, EnumMeshBufferPart part) => part switch + { + EnumMeshBufferPart.Xyz => _meshes.MappedPointer(meshId, MeshManager.BufferXyz), + EnumMeshBufferPart.Normals => _meshes.MappedPointer(meshId, MeshManager.BufferNormals), + EnumMeshBufferPart.Uv => _meshes.MappedPointer(meshId, MeshManager.BufferUv), + EnumMeshBufferPart.Rgba => _meshes.MappedPointer(meshId, MeshManager.BufferRgba), + EnumMeshBufferPart.Flags => _meshes.MappedPointer(meshId, MeshManager.BufferFlags), + EnumMeshBufferPart.CustomFloats => _meshes.MappedPointer(meshId, MeshManager.BufferCustomFloat), + EnumMeshBufferPart.CustomShorts => _meshes.MappedPointer(meshId, MeshManager.BufferCustomShort), + EnumMeshBufferPart.CustomInts => _meshes.MappedPointer(meshId, MeshManager.BufferCustomInt), + EnumMeshBufferPart.CustomBytes => _meshes.MappedPointer(meshId, MeshManager.BufferCustomByte), + _ => _meshes.MappedPointer(meshId, -1), + }; + + public void DeleteMesh(int meshId) => _meshes.Delete(meshId, _frames); +} + +/// Which draw command a native draw recorded, so the stats separate the kinds. +internal enum NativeDrawKind : byte +{ + /// The three-vertex fullscreen triangle a post pass generates in the shader. + Fullscreen = 0, + /// One indexed or non-indexed draw of one mesh: sky, entities, GUI quads. + Mesh = 1, + /// One mesh drawn with per-instance attributes: the particle pools. + Instanced = 2, + /// One indirect multi-draw out of the per-slot indirect ring: chunk pools and decals. + Indirect = 3, +} + +/// +/// Mesh draws on the native device API (docs/vulkan.md, decision 4: +/// "fullscreen triangle, mesh, multi-draw or instanced"). +/// +/// Stage 1 recorded fullscreen draws only. World systems are mesh draws, so these four entry +/// points join on the same preparation +/// (BeginNativeDraw) and swap the draw command for the mesh manager's: +/// +/// - the mesh's own vertex and index buffers, bound by , never a +/// second mesh path of this API's own; +/// - the mesh's interned vertex layout, stated on the pipeline +/// () and part of the pipeline key, so +/// a mesh pipeline and a fullscreen pipeline are never the same cache entry; +/// - the real mesh id threaded into BindProgramSets, which is what lets a chunk's +/// storage-buffer vertex fetch and an entity's Animation block resolve per draw +/// instead of against the fullscreen path's hardcoded 0; +/// - multi-draw through the existing per-slot indirect ring (AllocateIndirect), the same +/// regions every multi-draw allocates, so the ring's bookkeeping has one owner. +/// +/// What pins it: NativeMeshDrawTests (the ring's use and the pipeline-key dimensions) and +/// NativeSkyTests (the first ported system, old route against native route). +/// +public sealed unsafe partial class VulkanDevice +{ + /// + /// The topology a mesh was uploaded with, as the primitive a native pipeline rasterizes it + /// as. A mesh carries its own EnumDrawMode from the tesselator (triangles for most + /// geometry, lines for the aiming reticle, a line strip for the camera path), so a native + /// system states it from the mesh, as GL takes it from the VAO. Triangles for a mesh that + /// does not exist, so a caller that has already been refused a pipeline sees no surprise. + /// + internal PrimitiveTopology NativeMeshTopology(int meshId) => + GlEnums.TopologyFrom(_meshes.Get(meshId)?.DrawMode ?? Vintagestory.API.Client.EnumDrawMode.Triangles); + + /// Counts one native draw, once in the total and once in its own kind. + private void NoteNativeDraw(NativeDrawKind kind) + { + VulkanStats.NoteNativeDraw(); + // The generic stated route is counted apart, so the counters below say what the dedicated + // routes recorded (the differential tests compare the two). + if (_nativePass is { Generic: true }) + { + return; + } + _nativeDraws++; + switch (kind) + { + case NativeDrawKind.Fullscreen: + _nativeFullscreenDraws++; + VulkanStats.NoteNativeFullscreenDraw(); + break; + case NativeDrawKind.Mesh: + _nativeMeshDraws++; + VulkanStats.NoteNativeMeshDraw(); + break; + case NativeDrawKind.Instanced: + _nativeInstancedDraws++; + VulkanStats.NoteNativeInstancedDraw(); + break; + case NativeDrawKind.Indirect: + _nativeIndirectDraws++; + VulkanStats.NoteNativeIndirectDraw(); + break; + } + } + + /// + /// One indexed draw of one mesh: the sky dome, an entity shape, a GUI quad. The OpenGL body + /// is ClientPlatformWindows.RenderMesh(MeshRef). + /// + internal bool DrawNativeMesh(NativePipeline pipeline, int meshId, ReadOnlySpan textures) => + DrawNativeMeshInstanced(pipeline, meshId, 1, textures); + + /// + /// One indexed draw of one mesh with instances, the + /// per-instance attributes coming from the mesh's own instanced bindings (the particle + /// pools). The OpenGL body is ClientPlatformWindows.RenderMeshInstanced. + /// + internal bool DrawNativeMeshInstanced(NativePipeline pipeline, int meshId, int instanceCount, + ReadOnlySpan textures) + { + if (instanceCount <= 0) return false; + if (!NativeMeshIsDrawable(pipeline, meshId, indexed: true, out VulkanMesh? mesh)) return false; + if (!BeginNativeDraw(pipeline, textures, meshId, out CommandBuffer commandBuffer, out VulkanFramebuffer? target)) + { + return false; + } + + Checkpoint(commandBuffer, + CheckpointMarker.Draw(CheckpointKind.Draw, pipeline.ProgramId, target!.Id, meshId)); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native mesh=" + meshId + " program=" + pipeline.ProgramId + " pass='" + + _nativePass!.Name + "' target=" + target.Id + " indices=" + mesh!.IndexCount + + " instances=" + instanceCount); + } + + _meshes.Bind(commandBuffer, mesh!); + _context.Api.CmdDrawIndexed(commandBuffer, (uint)mesh!.IndexCount, (uint)instanceCount, 0, 0, 0); + NoteNativeDraw(instanceCount > 1 ? NativeDrawKind.Instanced : NativeDrawKind.Mesh); + return true; + } + + /// + /// One non-indexed draw of a mesh's vertex buffers: vertices, + /// instances, no index buffer. GL's glDrawArrays, for a + /// system whose geometry carries no index array. + /// + internal bool DrawNativeMeshArrays(NativePipeline pipeline, int meshId, int vertexCount, int instanceCount, + ReadOnlySpan textures) + { + if (vertexCount <= 0 || instanceCount <= 0) return false; + if (!NativeMeshIsDrawable(pipeline, meshId, indexed: false, out VulkanMesh? mesh)) return false; + if (!BeginNativeDraw(pipeline, textures, meshId, out CommandBuffer commandBuffer, out VulkanFramebuffer? target)) + { + return false; + } + + Checkpoint(commandBuffer, + CheckpointMarker.Draw(CheckpointKind.Draw, pipeline.ProgramId, target!.Id, meshId)); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native mesh arrays=" + meshId + " program=" + pipeline.ProgramId + " pass='" + + _nativePass!.Name + "' target=" + target.Id + " vertices=" + vertexCount + + " instances=" + instanceCount); + } + + _meshes.Bind(commandBuffer, mesh!); + _context.Api.CmdDraw(commandBuffer, (uint)vertexCount, (uint)instanceCount, 0, 0); + NoteNativeDraw(instanceCount > 1 ? NativeDrawKind.Instanced : NativeDrawKind.Mesh); + return true; + } + + /// + /// The multi-draw one mesh pool issues per pass - every surviving range of a chunk pool or + /// the decal pool in one command - through the existing per-slot indirect ring. The OpenGL + /// body is ClientPlatformWindows.RenderMesh(MeshRef, int[], int[], int) (glMultiDrawElements). + /// + /// holds GL's 64-bit byte offsets as pairs of ints, as + /// MeshDataPool passes them; is the + /// one place that converts them. + /// + internal bool DrawNativeMeshMulti(NativePipeline pipeline, int meshId, int[] indicesStarts, int[] indicesSizes, + int groupCount, ReadOnlySpan textures) + { + if (groupCount <= 0 || indicesStarts == null || indicesSizes == null) return false; + if (!NativeMeshIsDrawable(pipeline, meshId, indexed: true, out VulkanMesh? mesh)) return false; + if (!BeginNativeDraw(pipeline, textures, meshId, out CommandBuffer commandBuffer, out VulkanFramebuffer? target)) + { + return false; + } + + Checkpoint(commandBuffer, + CheckpointMarker.Draw(CheckpointKind.DrawMulti, pipeline.ProgramId, target!.Id, meshId)); + + VulkanBuffer indirect = AllocateIndirect(groupCount, out ulong indirectOffset); + if (RenderTrace.Enabled) + { + RenderTrace.Write("native multidraw mesh=" + meshId + " program=" + pipeline.ProgramId + " pass='" + + _nativePass!.Name + "' target=" + target.Id + " groups=" + groupCount + + " indirectOffset=" + indirectOffset); + } + + _meshes.DrawMulti(commandBuffer, meshId, indicesStarts, indicesSizes, groupCount, indirect, indirectOffset); + NoteNativeDraw(NativeDrawKind.Indirect); + return true; + } + + /// + /// Whether the mesh exists, carries what the draw needs, and is the shape the pipeline was + /// built for. The layout check is the one that matters: a pipeline built for another mesh's + /// layout would read attributes out of buffers that are not there, which no validation layer + /// can see because the descriptors are all valid. + /// + private bool NativeMeshIsDrawable(NativePipeline pipeline, int meshId, bool indexed, out VulkanMesh? mesh) + { + mesh = _meshes.Get(meshId); + if (mesh == null) + { + if (RenderTrace.Enabled) RenderTrace.Write("native draw skipped: no mesh " + meshId); + return false; + } + if (indexed && (mesh.Indices == null || mesh.IndexCount == 0)) + { + if (RenderTrace.Enabled) RenderTrace.Write("native draw skipped: mesh " + meshId + " has no indices"); + return false; + } + if (mesh.LayoutId != pipeline.Description.VertexLayoutId) + { + AddDiagnostic("native draw of mesh " + meshId + " (vertex layout " + mesh.LayoutId + + ") through a pipeline built for vertex layout " + pipeline.Description.VertexLayoutId); + return false; + } + return true; + } +} diff --git a/Optimum.Render.Vulkan/VulkanDevice.cs b/Optimum.Render.Vulkan/VulkanDevice.cs new file mode 100644 index 00000000..6764fdd0 --- /dev/null +++ b/Optimum.Render.Vulkan/VulkanDevice.cs @@ -0,0 +1,1362 @@ +using System; +using System.Collections.Generic; +using Optimum.Render.Vulkan.Core; +using Optimum.Render.Vulkan.Shaders; +using Silk.NET.Vulkan; +using Vintagestory.API.Client; +using Vintagestory.API.Config; + +using Buffer = Silk.NET.Vulkan.Buffer; + +namespace Optimum.Render.Vulkan; + +/// +/// The Vulkan renderer behind , which owns it +/// and calls it from every graphics override (Vulkan-native plan, Phase 1A). +/// +/// It accepts the game's stateful rendering contract - state, named uniforms, +/// texture bindings and mesh draws - and records native Vulkan passes. Mods using +/// the client API can take this route; direct OpenGL calls need separate support. +/// +/// Managers own their resources and expose integer ids for the client API. The +/// partial files group program, resource, mesh, binding and readback operations; +/// this facade coordinates frame submission and teardown. +/// +public sealed unsafe partial class VulkanDevice : IDisposable, Platform.ILatencyStageListener +{ + void Platform.ILatencyStageListener.OnFrameRenderStart() => NoteRenderStageStarted(); + + internal FrameTimingRecorder Latency { get; private set; } = new(); + internal ulong LatencyFrameId => _latencyFrameId; + private ulong _latencyFrameId; + private bool _latencyFrameIdPending; + private ulong _latencyRenderStartFrame; + + private void InitializeFrameTiming() + { + Latency = new FrameTimingRecorder(MirrorValidationMessage); + VulkanStats.LatencySource = Latency; + } + + /// Called before input; BeginFrame supplies an identity for headless callers. + public ulong BeginLatencyFrame() + { + _latencyFrameIdPending = true; + return ++_latencyFrameId; + } + + private void BeginLatencyFrameIdentity() + { + if (!_latencyFrameIdPending) BeginLatencyFrame(); + _latencyFrameIdPending = false; + _frames.Latency.FrameId = _latencyFrameId; + } + + internal void NoteRenderStageStarted() + { + if (_latencyRenderStartFrame == _latencyFrameId) return; + _latencyRenderStartFrame = _latencyFrameId; + Latency.Marker(_latencyFrameId, LatencyMarker.SimulationEnd); + Latency.Marker(_latencyFrameId, LatencyMarker.RenderSubmitStart); + } + + private void DisposeLatency() + { + if (ReferenceEquals(VulkanStats.LatencySource, Latency)) VulkanStats.LatencySource = null; + } + + private VulkanContext _context = null!; + private UploadManager _uploads = null!; + private TextureManager _textures = null!; + private MeshManager _meshes = null!; + private RenderTargetManager _targets = null!; + + /// + /// The streaming frame graph (Phase 2 step 2). On unless OPTIMUM_VULKAN_FRAMEGRAPH=0; + /// off, declarations only bind and every scope comes from inference as before. + /// + private readonly Graph.FrameGraph _graph = new(); + private GraphicsPipelineCache _pipelines = null!; + + /// Compute programs and their pipelines (the frame graph's compute pass kind). + private ComputePipelineCache _compute = null!; + + /// Per-slot descriptor sets of compute passes, reset when the slot begins a frame. + private ComputeDescriptorArena[] _computeArenas = Array.Empty(); + private DescriptorCache _descriptors = null!; + + /// How many descriptor sets the cache currently holds. For tests. + internal int CachedDescriptorSets => _descriptors.Count; + private FrameRing _frames = null!; + private ShaderCompiler _shaderCompiler = null!; + + /// The native shaders loaded at device start; null when they are off (docs/vulkan.md). + private NativeShaderLibrary? _nativeShaders; + private int _nativeLinks, _rewrittenLinks, _failedNativeLinks; + private int _nativeLinksReported, _rewrittenLinksReported, _failedNativeLinksReported; + + /// + /// Where compiled SPIR-V and the driver's pipeline cache are kept between launches, + /// set before . Null keeps nothing, which is what tests get; + /// the platform points it at the game's per-user cache folder. OPTIMUM_VULKAN_SHADER_CACHE + /// overrides it: a path to use instead, or 0 to keep nothing. + /// + public string? ShaderCacheDirectory { get; set; } + + /// + /// False forces the rewriter for every program, as OPTIMUM_VK_NATIVE_SHADERS=0 does; null follows the + /// environment. Read at . + /// + public bool? NativeShadersEnabled { get; set; } + + /// + /// The directory holding shaders.manifest.json, in place of shaders-vk beside the renderer + /// assembly (and of OPTIMUM_VK_SHADER_SOURCE). Read at . For tests. + /// + internal string? NativeShaderDirectory { get; set; } + + /// + /// The launcher's mod shader scan (OptimumConfig.IsShaderProgramOverriddenByMods): true for a pass name + /// whose GLSL a mod replaced, and for "all" when every program is rewriter-only. Such a program links + /// through the rewriter from the mod's source instead of the native SPIR-V. Null consults no scan (tests, a + /// device outside the client); OPTIMUM_VK_NATIVE_SHADERS=force ignores it. Read at . + /// + public Func? ShaderProgramOverriddenByMods { get; set; } + + /// True ignores as OPTIMUM_VK_NATIVE_SHADERS=force does; null follows the environment. For tests. + internal bool? IgnoreModShaderScan { get; set; } + + /// The scan consults: null when there is none or it is ignored. + private Func? _modShaderScan; + private readonly HashSet _modOverrideLogged = new(StringComparer.OrdinalIgnoreCase); + + /// What made of the native shaders: the origin and program count, or why they are off. + internal string NativeShaderStatus { get; private set; } = "not loaded"; + + /// Programs linked from the manifest, through the rewriter, and native programs that fell back, since the device came up. + internal (int Native, int Rewritten, int Failed) ShaderLinkCounts => (_nativeLinks, _rewrittenLinks, _failedNativeLinks); + + /// Whether a linked program came from the native manifest. + internal bool IsNativeProgram(int programId) => + _programs.TryGetValue(programId, out ShaderProgramResources? program) && program.IsNative; + + /// The pipeline cache and key-log files and their saves; null when there is no cache root. + private PipelineCachePersistence? _pipelinePersistence; + + /// + /// Forces blocking pipeline creation (every draw lands in the frame that issues it) when + /// true, allows background compiles with skipped draws when false; null reads + /// OPTIMUM_VULKAN_SYNC_PIPELINES. Set before . GPU tests that read + /// pixels back after one frame get true from GpuTest. + /// + public bool? SynchronousPipelines { get; set; } + + internal static bool ResolveSynchronousPipelines(bool? configured, string? environment) => + configured ?? environment?.Trim() is "1" or "on" or "true"; + + /// + /// A parity capture forces blocking creation so its exact frame includes every draw. + /// An explicit still wins. + /// + internal static bool ResolveSynchronousPipelines(bool? configured, string? environment, string? parityDump) => + configured ?? (ResolveSynchronousPipelines(null, environment) || NamesCaptureDirectory(parityDump)); + + private static bool NamesCaptureDirectory(string? value) => + !string.IsNullOrWhiteSpace(value) && System.IO.Path.IsPathRooted(value); + + /// The pipeline cache. Tests only. + internal GraphicsPipelineCache PipelinesForTests => _pipelines; + + internal static string? ResolveShaderCacheRoot(string? configured, string? environment) + { + if (string.IsNullOrWhiteSpace(environment)) return string.IsNullOrWhiteSpace(configured) ? null : configured; + string value = environment.Trim(); + return value is "0" or "off" or "false" ? null : value; + } + + private readonly Dictionary _programs = new(); + + /// Pass names by program id, so a device-loss report can name the shader. + private readonly Dictionary _programNames = new(); + + /// Scratch for the SSBO path's pruned custom ints, grown as needed. + private int[] _prunedCustomInts = []; + + private uint _frameCounter; + private uint _uniformExhaustionReportedFrame = uint.MaxValue; + private readonly Dictionary _stagedStages = new(); + + /// + /// Error-severity diagnostics since the last GetError, under their own lock: + /// the layers call back from whichever thread made the Vulkan call. + /// + private readonly List _errors = new(); + + /// + /// How many entries holds. GetError runs after every + /// render stage, and in steady state this read is all it costs. + /// + private volatile int _errorCount; + + /// A client that never drains the queue does not grow it without bound. + private const int MaxQueuedErrors = 1024; + + /// What the frame command buffer already holds, so a draw emits only changed dynamic state. + private readonly DynamicStateCache _dynamicState = new(); + private long _dynamicStateCommands; + + /// Which resources are young enough that their descriptor sets belong in the per-slot arena. + private readonly ResourceAge _resourceAge = new(); + private DescriptorArena[] _descriptorArenas = Array.Empty(); + + /// Per-slot indirect-command buffers; see . + private IndirectRing _indirectRing = null!; + private VulkanBuffer?[] _indirectBuffers = Array.Empty(); + + /// Buffers taken this frame by multi-draws that did not fit their slot's buffer. + private readonly List _indirectOverflow = new(); + private ulong _indirectOverflowCursor; + private long _indirectOverflows; + private long _indirectGrowths; + + /// + /// Decision 9's set 1 and the one shared pipeline layout every program's pipelines + /// are built against. Created at bring-up and kept current (slots retire with their + /// textures, writes flush before every submission). + /// + private BindlessTextureTable? _bindless; + private SharedPipelineLayout? _sharedLayout; + + /// + /// The shared frame block (set 0, ): the CPU shadow every + /// frame-global write lands in, and the ring snapshot draws bind until a write + /// changes something. Replaces up to 56 per-program copies of the same values. + /// + private readonly byte[] _frameGlobals = FrameGlobals.CreateShadow(); + private uint _frameGlobalsVersion = 1; + private uint _frameGlobalsSnapshotFrame; + private uint _frameGlobalsSnapshotVersion; + private uint _frameGlobalsSnapshotOffset; + + /// Sixteen bytes of float defaults followed by sixteen of int. + private const ulong DefaultAttributeBufferSize = 32; + + private VulkanBuffer? _defaultAttributes; + + /// + /// The zero-filled buffer at every set 2 binding a draw has nothing for. An unbound + /// sampler reads the bindless table's placeholder of its kind instead. + /// + private VulkanBuffer? _placeholderUniforms; + + // Atlas composition reads one tile while writing another in the same image. + // Each such draw takes a pooled ReadSelf copy, refreshed before the draw and + // released when the next draw's samplers are resolved (Phase 2 step 4). + private Graph.FeedbackCopyPool _readSelfCopies = null!; + private readonly Dictionary _sampledTextureOverrides = new(); + + // Physical backing for frame-graph transients (Transient pool class). + private Graph.TransientAllocator _transients = null!; + + private int _nextProgramId = 1; + private bool _frameActive; + private bool _disposed; + + /// A stage that has been preprocessed but not yet linked. + private sealed class StagedStage + { + public EnumShaderType Stage; + public string Code = ""; + public string PrefixCode = ""; + public string Filename = "shader"; + } + + // ------------------------------------------------------------------ lifecycle + + public string BackendName => "Vulkan"; + + /// + /// Whether this machine can run the backend, decided without a window. + /// + /// The check has to happen before the window is created, because a window + /// opened with no graphics API cannot be handed back to OpenGL without being + /// destroyed and reopened. Creating an instance and a device is the only + /// honest way to know - driver support for the required 1.3 features is not + /// something that can be inferred from a vendor string. + /// + /// + /// Whether OPTIMUM_VULKAN_VALIDATION asks for the validation layers. + /// + /// Read once: the answer cannot change within a process, because the layers + /// are baked into the instance. + /// + private static readonly string? ValidationSetting = + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_VALIDATION"); + + private static readonly bool ValidationRequestedByEnvironment = + !string.IsNullOrEmpty(ValidationSetting); + + /// + /// Where a bare OPTIMUM_VULKAN_VALIDATION=1 mirrors the layer's messages. + /// Before this default the messages only surfaced when the client happened + /// to poll the error channel, and a whole class of hazards went unlogged. + /// + private static readonly string DefaultValidationLogPath = + System.IO.Path.Combine(System.IO.Path.GetTempPath(), "optimum-vulkan-validation.log"); + + /// + /// Where validation messages are mirrored, when the variable names a path + /// rather than just switching the layers on. + /// + /// The client's own error channel only surfaces them when it happens to call + /// CheckGlError, and a device-lost kills the process before that; a file gets + /// the message that preceded the loss. + /// + private static readonly string? ValidationLogPath = + ResolveValidationLogPath(ValidationSetting, DefaultValidationLogPath); + + /// + /// A setting that names a path is used as one; anything else (the bare "1") + /// only switches the layers on and mirrors to . + /// Windows separators count as a path too, so "C:\logs\vulkan.log" is not + /// silently redirected to the temp file. + /// + internal static string? ResolveValidationLogPath(string? setting, string fallback) + { + if (setting == null) return null; + return setting.Contains('/') || setting.Contains('\\') ? setting : fallback; + } + + /// + /// OPTIMUM_VULKAN_VALIDATION_FEATURES: comma list of "sync" (synchronization + /// validation), "best" (best practices with the NVIDIA and AMD sets), "mobile" + /// (the Arm and IMG sets), "gpu" (GPU-assisted) and "gpu-only" (GPU-assisted, + /// core off). Requested through VK_EXT_layer_settings (the deprecated + /// VK_EXT_validation_features on older layers) so it does not depend on the + /// layer's environment variable names, which changed. + /// + private static readonly string ValidationFeatureSetting = + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_VALIDATION_FEATURES") ?? ""; + + /// + /// The client logs diagnostics through string.Format, and a layer message + /// that prints a struct ("pImageMemoryBarriers[0]: { ... }") throws a + /// FormatException there and is lost. Braces become brackets before the + /// message reaches either channel. + /// + private static string SanitiseForClientLog(string message) => + message.Replace('{', '[').Replace('}', ']'); + + private static void MirrorValidationMessage(string message) + { + if (ValidationLogPath == null) return; + try + { + System.IO.File.AppendAllText(ValidationLogPath, message + "\n"); + } + catch (Exception error) when ( + error is System.IO.IOException + or UnauthorizedAccessException + or NotSupportedException + or ArgumentException) + { + // A diagnostic write must never take the device down: a read-only + // directory or a malformed path is a lost log line, nothing more. + } + } + + public static bool IsSupported(out string failureReason) + { + string driver; + return IsSupported(false, out failureReason, out driver); + } + + /// + /// Probes for a usable device, and on the "auto" setting also decides whether + /// this driver is one the backend is trusted on. + /// + /// An explicit "vulkan" means the user asked for it and gets it wherever it + /// runs at all. "auto" is the setting a player never chose, so it takes the + /// backend only on driver families it has actually been exercised against; + /// everything else stays on OpenGL, which is the path that certainly works. + /// The list is deliberately about the driver rather than the GPU model: + /// behaviour that breaks a backend lives in the driver. + /// + public static bool IsSupported(bool automatic, out string failureReason, out string driverName) + { + driverName = "unknown"; + + var options = new VulkanContextOptions { Headless = true }; + if (!VulkanContext.TryCreate(options, out VulkanContext? context, out string? reason)) + { + failureReason = reason ?? "no usable Vulkan device"; + return false; + } + + try + { + driverName = context!.Capabilities.DriverName ?? "unknown"; + if (automatic && !IsAllowedForAutomaticSelection(driverName)) + { + failureReason = "the automatic setting does not select Vulkan on this driver (" + + driverName + "); set Renderer to \"vulkan\" to use it anyway"; + return false; + } + } + finally + { + context!.Dispose(); + } + + failureReason = null!; + return true; + } + + /// + /// The driver families the backend is regularly run against. Matching is on a + /// substring of the reported driver name, because vendors version the rest of + /// the string freely. + /// + private static readonly string[] AutomaticSelectionAllowList = + { + // Exercised continuously during development, both test suite and client. + "NVIDIA", + // Mesa's Intel driver, which is also what the Arc target uses on Linux. + "Intel open-source Mesa driver", + "Mesa", + // Windows Intel driver, the Claw's own. + "Intel Corporation", + // Mesa's AMD driver. + "radv", + "AMD proprietary driver", + }; + + internal static bool IsAllowedForAutomaticSelection(string driverName) + { + foreach (string allowed in AutomaticSelectionAllowList) + { + if (driverName.Contains(allowed, StringComparison.OrdinalIgnoreCase)) return true; + } + return false; + } + + public bool Initialize(IntPtr windowHandle, int width, int height, out string failureReason) + { + bool headless = windowHandle == IntPtr.Zero; + + var options = new VulkanContextOptions + { + // A window handle of zero means no presentation surface, which is how + // capability probes and tests bring the device up. + Headless = headless, + // Validation layers are chosen when the instance is created, so the + // client's GlDebugMode setting is too late to turn them on - it is + // applied to DebugMode only after the device exists. OPTIMUM_VULKAN_VALIDATION + // is the way to get them for a real client session, which is the + // only place the world-loading paths actually run. + EnableValidation = DebugMode || ValidationRequestedByEnvironment, + ValidationFeatures = ValidationFeatureSetting, + DebugCallback = message => + { + AddDiagnostic(SanitiseForClientLog(message)); + MirrorValidationMessage(message); + if (RenderTrace.Enabled) + RenderTrace.Write("validation: target=" + (_targets?.Bound?.Id ?? -1) + " " + message); + }, + // Surface extensions have to be enabled at instance creation, before + // any surface can exist, so the window system is asked first. + RequiredInstanceExtensions = headless + ? Array.Empty() + : WindowSurface.RequiredInstanceExtensions(), + }; + + ConfigureContextOptions?.Invoke(options); + + if (!VulkanContext.TryCreate(options, out VulkanContext? context, out failureReason)) + { + return false; + } + + _context = context!; + // Any hard Vulkan failure now reaches the client's error channel and the + // validation log instead of turning into a silent stall. + VulkanResult.OnFailure = message => + { + // A failed Vulkan call is an error by definition, so it carries the + // same prefix the layers' error-severity messages do and reaches the + // client through GetError. + AddDiagnostic(VulkanContext.ErrorPrefix + message); + MirrorValidationMessage(message); + }; + VulkanResult.DescribeDeviceLoss = DescribeDeviceLoss; + MirrorValidationMessage("--- device up on " + _context.Capabilities.DeviceName + + "; validation layers " + (_context.ValidationEnabled ? "ENABLED " + _context.ValidationLayerVersion : "NOT AVAILABLE") + + (_context.ValidationSettingsApplied.Length == 0 ? "" : "; " + _context.ValidationSettingsApplied) + + "; GPU checkpoints " + (_context.CheckpointsAvailable ? "ENABLED" : "NOT AVAILABLE") + + "; device fault reporting " + (_context.DeviceFaultAvailable ? "ENABLED" : "NOT AVAILABLE") + + "; poison " + (_context.PoisonFreshResources ? "ON" : "off") + + "; color write tier " + DeviceCaps.Token(_context.Capabilities.ColorWriteTier) + + (_context.Capabilities.DynamicColorBlend ? " (dynamic blend)" : "") + + "; bindless sampled images per stage " + + _context.Capabilities.DescriptorIndexing.MaxPerStageDescriptorUpdateAfterBindSampledImages + + " (needs " + DescriptorIndexingFloor.RequiredSampledImages + ")" + + "; push constants " + _context.Capabilities.DescriptorIndexing.MaxPushConstantsSize + " B" + + "; " + _context.Capabilities.LatencySummary); + // Seams S1-S5: the backend is installed before the frame ring and the first + // swapchain exist, so nothing in the frame ever sees a different instance. + InitializeFrameTiming(); + // A ReBAR miss is logged, not an error: the validation mirror and the + // trace, never GetError. The stats sample reads this allocator's heaps. + _context.Allocator.Log = MirrorValidationMessage; + VulkanStats.MemorySource = _context.Allocator; + // Uploads never wait: they ride the next frame submission, recorded from + // any thread into the ring's upload batch (or inline into the frame when + // it already used the destination; see UploadManager). + _frames = new FrameRing(_context); + // Seam S4: the installed backend tags this ring's submits. + _uploads = _frames.Uploads; + _textures = new TextureManager(_context, _uploads); + _meshes = new MeshManager(_context, _uploads); + _targets = new RenderTargetManager(_context, _textures, _graph); + // An inline upload records transfer commands into the frame command + // buffer, which no rendering scope may enclose. + _uploads.CloseRenderingScope = commandBuffer => _targets.EndRendering(commandBuffer); + // A barrier flushed into the frame command buffer while a scope is open + // is a transition inside the scope; debug builds reject it. + _textures.ScopeOpen = commandBuffer => + _frameActive && _targets.RenderingActive && commandBuffer.Handle == Commands.Handle; + _barriers = _textures.CreateBatcher(); + // Transients and ReadSelf copies live in the Transient pool class; both + // release through ReleaseTexture, which retires on the timeline. + _transients = new Graph.TransientAllocator(new Graph.TextureTransientBacking(_textures, ReleaseTexture), + TransientAliasingOverride ?? Graph.TransientAllocator.AliasingFromEnvironment()); + _readSelfCopies = new Graph.FeedbackCopyPool(_frames.Timeline, CreateReadSelfCopy, ReleaseTexture); + string? cacheRoot = ResolveShaderCacheRoot(ShaderCacheDirectory, + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_SHADER_CACHE")); + byte[]? pipelineSeed = null; + if (cacheRoot != null) + { + _pipelinePersistence = PipelineCachePersistence.Open(cacheRoot, + PipelineCacheIdentity.Of(_context.Capabilities), out pipelineSeed); + _pipelinePersistence.Log = MirrorValidationMessage; + } + _pipelines = new GraphicsPipelineCache(_context, _context.Capabilities.ColorWriteTier, + _context.Capabilities.DynamicColorBlend, pipelineSeed); + // Background compiles (docs/vulkan.md#caches, design item 4): a draw whose + // pipeline is not in the driver cache is skipped while a worker compiles it. + bool synchronousPipelines = ResolveSynchronousPipelines(SynchronousPipelines, + Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_SYNC_PIPELINES"), + Environment.GetEnvironmentVariable("OPTIMUM_PARITY_DUMP")); + _pipelines.AsyncCompiles = !synchronousPipelines; + _pipelines.KeyLog = _pipelinePersistence?.KeyLog; + _descriptors = new DescriptorCache(_context); + _compute = new ComputePipelineCache(_context, () => _pipelines.DriverCache); + // Decision 9: the bindless table retires a texture's slots on the timeline + // values of its deletion, and the shared layout names the table's set layout. + _bindless = new BindlessTextureTable(_context, _textures, _frames.Timeline); + _textures.Deleted = texture => + { + _bindless.Release(texture.Id); + ForgetFrameTexture(texture.Id); + }; + _sharedLayout = new SharedPipelineLayout(_context, _bindless.Layout); + _descriptorArenas = new DescriptorArena[_frames.FramesInFlight]; + for (int i = 0; i < _descriptorArenas.Length; i++) _descriptorArenas[i] = new DescriptorArena(_context); + _computeArenas = new ComputeDescriptorArena[_frames.FramesInFlight]; + for (int i = 0; i < _computeArenas.Length; i++) _computeArenas[i] = new ComputeDescriptorArena(_context); + _indirectRing = new IndirectRing(_frames.FramesInFlight); + _indirectBuffers = new VulkanBuffer?[_frames.FramesInFlight]; + _queryRing = new QueryRing(_context, _frames.Timeline, _frames.FramesInFlight); + _readbacks = new ReadbackManager(_context, _textures, _frames); + // A GL query counts across scope ends; a Vulkan one must not be active + // across vkCmdEndRendering, so the ring suspends and resumes it. + _targets.ScopeClosing = _queryRing.OnScopeClosing; + _targets.ScopeClosed = _queryRing.OnScopeClosed; + _targets.ScopeOpened = _queryRing.OnScopeOpened; + _shaderCompiler = new ShaderCompiler + { + BinaryCache = cacheRoot == null ? null : new ShaderBinaryCache(System.IO.Path.Combine(cacheRoot, "spirv")), + }; + LoadNativeShaders(); + MirrorValidationMessage(cacheRoot == null + ? "--- shader cache off" + : "--- shader cache " + cacheRoot + "; pipeline cache " + + (pipelineSeed == null ? "cold" : _pipelines.SeedAccepted ? "warm (" + pipelineSeed.Length + " bytes)" : "rejected by the driver") + + "; pipeline key log " + (_pipelinePersistence?.KeyLog.Count ?? 0) + " entries"); + MirrorValidationMessage("--- pipelines " + (_pipelines.AsyncCompiles + ? "compile in the background" + : synchronousPipelines ? "compile blocking (OPTIMUM_VULKAN_SYNC_PIPELINES, a frame capture or the device setting)" : "compile blocking (no pipelineCreationCacheControl)")); + CreateDefaultAttributeBuffer(); + CreatePlaceholderUniformBuffer(); + + // The default target is an ordinary offscreen one, so it exists headless too: the + // frame's last passes (the blit) write into it, and a headless run - the capture + // harness, a test driving the post chain - reads it back. Only presenting it needs + // a surface. + CreateDefaultFramebuffer((uint)width, (uint)height); + + if (!headless) + { + if (!WindowSurface.TryCreate(_context, windowHandle, out SurfaceKHR surface, out string? surfaceError)) + { + failureReason = surfaceError ?? "could not create a presentation surface"; + return false; + } + + if (!Swapchain.TryCreate(_context, surface, (uint)width, (uint)height, _vsync, _frames.Timeline, + out Swapchain? swapchain, out string? swapchainError)) + { + failureReason = swapchainError ?? "could not create a swapchain"; + return false; + } + + _swapchain = swapchain; + // Seam S2: VkPresentIdKHR may only be chained when VK_KHR_present_id and its + // feature were actually enabled; chaining it otherwise is a validation error. + _swapchain!.PresentIdEnabled = _context.Capabilities.PresentIdEnabled; + _presentPath = new BlitPresentPath(_context, _textures, DefaultColorTexture); + } + + failureReason = null!; + return true; + } + + private Swapchain? _swapchain; + private BlitPresentPath? _presentPath; + private readonly MissedVsyncDetector _missedVsyncs = new(); + private long _lastPresentReturn; + private string? _reportedRebuildFailure; + private bool _vsync = true; + + private VulkanTexture? DefaultColorTexture() => _textures.Get(_defaultColor); + private int _defaultFramebuffer; + private int _defaultColor; + private int _defaultDepth; + private uint _windowWidth; + private uint _windowHeight; + + /// + /// The target the client renders into when it asks for the default + /// framebuffer. + /// + /// It is an ordinary offscreen target rather than the swapchain image, + /// because the game reads it back for screenshots and because presenting is + /// where the one flip happens. Rendering straight into a swapchain image + /// would put that flip in the middle of the pipeline. + /// + private void CreateDefaultFramebuffer(uint width, uint height) + { + _windowWidth = Math.Max(width, 1); + _windowHeight = Math.Max(height, 1); + + _defaultColor = _textures.Create(_windowWidth, _windowHeight, Format.R8G8B8A8Unorm); + _defaultDepth = _textures.Create(_windowWidth, _windowHeight, Format.D32Sfloat); + + _defaultFramebuffer = _targets.Create(_windowWidth, _windowHeight); + _targets.Attach(_defaultFramebuffer, 0, _defaultColor); + _targets.Attach(_defaultFramebuffer, -1, _defaultDepth); + + if (RenderTrace.Enabled) + { + RenderTrace.Write("default framebuffer id=" + _defaultFramebuffer + + " " + _windowWidth + "x" + _windowHeight + + " swapchain=" + (_swapchain == null + ? "none" + : _swapchain.Extent.Width + "x" + _swapchain.Extent.Height)); + } + } + + /// + /// Builds the buffer that stands in for GL's constant generic vertex + /// attribute, holding (0, 0, 0, 1) as floats and again as integers. + /// + /// GL guarantees that value for any attribute the draw does not supply, and + /// shaders here rely on it - a vertex flags word of zero means no glow and no + /// z-offset, a damage effect of zero means no discard. Vulkan has no such + /// default, so the value has to come from somewhere real: this buffer, bound + /// at a reserved binding with stride zero so every vertex reads it. + /// + private void CreateDefaultAttributeBuffer() + { + _defaultAttributes = new VulkanBuffer(_context, DefaultAttributeBufferSize, + BufferUsageFlags.VertexBufferBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + + var floats = new float[] { 0f, 0f, 0f, 1f }; + var integers = new int[] { 0, 0, 0, 1 }; + + fixed (float* source = floats) + { + System.Buffer.MemoryCopy(source, (void*)_defaultAttributes.Mapped, 16, 16); + } + fixed (int* source = integers) + { + System.Buffer.MemoryCopy(source, (void*)(_defaultAttributes.Mapped + 16), 16, 16); + } + } + + /// + /// Builds the zero-filled buffer that fills any shader-declared uniform block + /// the client has not supplied a buffer for yet, and every other set 2 binding + /// a draw does not read (the shared layout's set is written whole). + /// + /// Same reasoning as the placeholder texture: leaving the binding undefined + /// makes every draw with that program invalid, so a program whose UBO has not + /// been created yet would take the whole frame down rather than read zeroes. + /// GL reads zeroes from an unbacked block, so this is also the closer match. + /// + private void CreatePlaceholderUniformBuffer() + { + // Large enough for the blocks the game declares - the animation transform + // block is the biggest at a few tens of kilobytes - and clamped to what + // the device will actually let a descriptor address. + ulong size = Math.Min(65536UL, Math.Max(16384UL, _context!.Capabilities.MaxUniformBufferRange)); + + _placeholderUniforms = new VulkanBuffer(_context, size, + BufferUsageFlags.UniformBufferBit | BufferUsageFlags.StorageBufferBit, + MemoryPropertyFlags.HostVisibleBit | MemoryPropertyFlags.HostCoherentBit); + + if (_placeholderUniforms.Mapped != IntPtr.Zero) + { + new Span((void*)_placeholderUniforms.Mapped, (int)size).Clear(); + } + } + + private void DestroyDefaultFramebuffer() + { + if (_defaultFramebuffer > 0) _targets.Delete(_defaultFramebuffer); + if (_defaultColor > 0) ReleaseTexture(_defaultColor); + if (_defaultDepth > 0) ReleaseTexture(_defaultDepth); + + _defaultFramebuffer = 0; + _defaultColor = 0; + _defaultDepth = 0; + } + + public string RendererString => _context?.Capabilities.DeviceName ?? "Vulkan"; + public string VendorString => _context?.Capabilities.DriverName ?? "unknown"; + public string VersionString => _context == null + ? "unknown" + : VulkanContext.VersionString(_context.Capabilities.ApiVersion); + + /// + /// Reported as a GLSL version because the client parses it to decide whether + /// a shader's #version is supported. The backend accepts everything the + /// translator accepts, which is well above what any shader in the game asks + /// for. + /// + public string ShaderVersionString => "4.50"; + + public int MaxTextureSize => (int)(_context?.Capabilities.MaxImageDimension2D ?? 0); + public bool SupportsThickLines => _context?.Capabilities.WideLines ?? false; + public bool SupportsSSBOs => true; + + public bool DebugMode { get; set; } + + /// + /// Test seam: adjusts the context options builds, + /// just before the context is created (validation features, a message + /// recorder, poison mode). Null in the client. + /// + internal Action? ConfigureContextOptions { get; set; } + + /// + /// Drains queued diagnostics, reporting only what the layers called an + /// error. + /// + /// The client turns a non-null result into a thrown exception via + /// CheckGlError, so this has to mean "something is actually wrong" - the + /// GL call it stands in for, glGetError, never reported advice. Warnings are + /// still dropped into the trace for anyone reading it. + /// + public string GetError() + { + // Phase 1B step 6: a volatile read; the message is built only when there is one. + if (_errorCount == 0) return null!; + + lock (_errors) + { + if (_errors.Count == 0) return null!; + string joined = string.Join("\n", _errors); + _errors.Clear(); + _errorCount = 0; + return joined; + } + } + + /// + /// Queues a diagnostic for GetError. Only error-severity messages (the + /// ones) are kept: GetError never + /// reported anything else, and warnings already reach the trace and the + /// validation log where they are raised. Safe from any thread. + /// + private void AddDiagnostic(string message) + { + if (!message.StartsWith(VulkanContext.ErrorPrefix, StringComparison.Ordinal)) + { + // A native route's refusal is not raised anywhere else: the trace carries it. + if (RenderTrace.Enabled && !message.StartsWith("[", StringComparison.Ordinal)) RenderTrace.Write("diagnostic: " + message); + return; + } + + lock (_errors) + { + if (_errors.Count >= MaxQueuedErrors) return; + _errors.Add(message); + _errorCount = _errors.Count; + } + } + + /// Queues a diagnostic as the device's own sources do. Tests only. + internal void AddDiagnosticForTests(string message) => AddDiagnostic(message); + + // ---------------------------------------------------------------------- frame + + public void BeginFrame() + { + // A native pass never spans a frame boundary. + _nativePass = null; + _nativeTarget = null; + + // CPU frame interval: start of one frame to the start of the next, so it + // includes the Frame timeline pacing wait below and everything the client did. + long frameStart = System.Diagnostics.Stopwatch.GetTimestamp(); + if (_lastFrameStart != 0) + { + VulkanStats.NoteFrameInterval( + (frameStart - _lastFrameStart) * 1000.0 / System.Diagnostics.Stopwatch.Frequency); + } + _lastFrameStart = frameStart; + ReportShaderLoad(); + + // The safe point for pipelines the background worker finished, and for the + // opportunistic pipeline cache save (a timestamp check; the save runs on a worker). + _pipelines.PublishCompleted(); + _pipelinePersistence?.Tick(_pipelines, frameStart); + + // The frame that ended: its ReadSelf copies wait on the Frame value it + // recorded (taken before the ring reserves the next), and its transient + // leases and bindings end. + ReleaseReadSelfCopies(); + _readSelfCopies.EndFrame(); + VulkanStats.NoteTransientFrame(_transients.PhysicalBytes + _transients.OptedInBytes, + _transients.AliasedBytes, _transients.Leases.Count, _transients.AliasedLeaseCount, _readSelfCopies.Live); + _transients.BeginFrame(); + + // Seam S2: the frame's latency identity, before the ring hands out the slot. + BeginLatencyFrameIdentity(); + + FrameSlot slot = _frames.BeginFrame(); + // After the ring's wait and collection, before anything is recorded: freed + // bindless slots get their placeholder back and queued writes land. + _bindless?.BeginFrame(); + _readSelfCopies.Collect(); + _frameActive = true; + _frameCounter++; + Checkpoint(Commands, CheckpointMarker.FrameBegin(_frameCounter)); + + // The slot's previous frame has completed (FrameRing waited for it), so + // its indirect cursor and descriptor arena reset wholesale. + BeginIndirectFrame(slot.Index); + _descriptorArenas[slot.Index].Reset(); + _computeArenas[slot.Index].Reset(); + _resourceAge.NoteFrame(ResourceIds.Highest); + _dynamicState.Invalidate(); + + // The slot's previous frame has finished: its query results move to the + // host buffer before the pools reset, and its readback arena is free again. + _queryRing.BeginSlot(slot.Index, slot.CommandBuffer); + _readbacks.BeginSlot(slot.Index); + + // Sets naming resources deleted since last frame leave the cache now and + // are freed once the Frame timeline has passed every frame that could + // have bound them. + IDisposable? freedSets = _descriptors.CollectReleases(); + if (freedSets != null) _frames.DeferDeletion(freedSets); + + VulkanStats.NoteFrame(); + if (StatsLogPath != null && + VulkanStats.SampleIfDue(TimeSpan.FromSeconds(1)) is { } sample) + { + try + { + // One sample is several lines (see VulkanStats); the first keeps the original format. + System.IO.File.AppendAllText(StatsLogPath, sample + "\n"); + } + catch (System.IO.IOException) + { + } + } + } + + /// Stopwatch timestamp of the last BeginFrame, 0 before the first. + private long _lastFrameStart; + + /// Deferred destructions still waiting on the timelines. Tests only. + /// The frame ring's timelines. Tests only. + internal FrameTimeline TimelineForTests => _frames.Timeline; + + /// The frame ring's upload manager. Tests only. + /// Decision 9's set 1. Tests only. + internal BindlessTextureTable BindlessForTests => _bindless!; + + /// Decision 9's shared pipeline layout. Tests only. + // ------------------------------------------------------------------ compute + + /// The compute programs. Tests only. + internal ComputePipelineCache ComputeForTests => _compute; + + /// A compute program from compiled SPIR-V; see . + internal int CreateComputeProgram(ComputeProgramDescription description) => _compute.Create(description); + + /// + /// Compiles a native compute shader and creates its program; 0, with the compiler's + /// message in , when it does not compile. + /// + internal int CreateComputeProgram(string code, string name, ComputeSlot[] slots, uint pushConstantBytes = 0, + uint localSizeX = 8, uint localSizeY = 8) + { + ShaderCompileResult compiled = _shaderCompiler.CompileCompute(code, name); + if (!compiled.Success) + { + AddDiagnostic(VulkanContext.ErrorPrefix + "compute shader '" + name + "' failed to compile: " + compiled.Error); + return 0; + } + return _compute.Create(new ComputeProgramDescription + { + Name = name, + Spirv = compiled.Spirv, + Slots = slots, + PushConstantBytes = pushConstantBytes, + LocalSizeX = localSizeX, + LocalSizeY = localSizeY, + }); + } + + /// Deletes a compute program once no submitted frame can still bind it. + internal void DeleteComputeProgram(int programId) + { + ComputeProgram? program = _compute.Remove(programId); + if (program != null) _frames.DeferDeletion(program); + } + + /// + /// A texture compute passes store to: where the device can + /// store to and sample it, else a wider format of the same kind (RGBA8 last); + /// levels. The chosen format is the texture's + /// . + /// + internal int CreateStorageTexture(int width, int height, Format format, int mipLevels = 1) => + _textures.CreateStorage((uint)Math.Max(1, width), (uint)Math.Max(1, height), format, (uint)Math.Max(1, mipLevels)); + + /// A live texture's description (size, levels, chosen format, usage) for compute pass owners; null when it does not exist. + internal VulkanTexture? TextureOf(int textureId) => _textures.Get(textureId); + + private Graph.ComputeImageInfo? ComputeImageInfoOf(int textureId) => + _textures.Get(textureId) is { } texture + ? new Graph.ComputeImageInfo(texture.Width, texture.Height, texture.MipLevels, texture.Cube ? 6u : texture.Layers) + : null; + + private static readonly SamplerState ComputeNearest = new(Filter.Nearest, Filter.Nearest, SamplerMipmapMode.Nearest, + SamplerAddressMode.ClampToEdge, SamplerAddressMode.ClampToEdge, 0f, false, 1f, BorderColor.FloatOpaqueBlack, + Mipmapped: true); + + private static readonly SamplerState ComputeLinear = ComputeNearest with + { + MagFilter = Filter.Linear, + MinFilter = Filter.Linear, + }; + + /// + /// Records a compute pass into the frame (): + /// closes any open rendering scope, lands clears pending on its images, queues one + /// barrier per binding level range from its access and flushes them as one command, + /// binds the pipeline for the pass's specialization values and one descriptor set, + /// and records every dispatch. False, with the reason in , + /// when no frame is open or the declaration does not fit its program. + /// + internal bool RecordComputePass(Graph.ComputePassDeclaration pass) + { + if (!_frameActive) return false; + + ComputeProgram? program = _compute.Get(pass.ProgramId); + string? error = program == null + ? "no compute program " + pass.ProgramId + : Graph.ComputePassPlanner.Validate(pass, ComputeImageInfoOf); + if (error == null && program != null) + { + foreach (Graph.ComputeBinding binding in pass.Bindings) + { + if (!program.TryGetSlot(binding.Binding, out ComputeSlot slot)) + { + error = "binding " + binding.Binding + " is not in program '" + program.Name + "'"; + break; + } + bool storage = Graph.ComputePassPlanner.IsStorage(binding.Access); + if (storage != (slot.Kind == ComputeSlotKind.Storage)) + { + error = "binding " + binding.Binding + " is declared " + slot.Kind + " by program '" + program.Name + + "' but bound " + binding.Access; + break; + } + if (storage && (_textures.Get(binding.TextureId)!.Usage & ImageUsageFlags.StorageBit) == 0) + { + error = "binding " + binding.Binding + " stores to texture " + binding.TextureId + + ", which was not created as a storage texture"; + break; + } + } + foreach (ComputeSlot slot in program.Slots) + { + if (error != null) break; + if (Array.FindIndex(pass.Bindings, b => b.Binding == slot.Binding) < 0) + error = "program '" + program.Name + "' binding " + slot.Binding + " is not bound"; + } + foreach (Graph.ComputeDispatch dispatch in pass.Dispatches) + { + if (error != null) break; + if (dispatch.PushConstants is { Length: > 0 } push && push.Length > program.PushConstantBytes) + error = "a dispatch pushes " + push.Length + " bytes; program '" + program.Name + "' declares " + + program.PushConstantBytes; + } + } + if (error != null) + { + AddDiagnostic(VulkanContext.ErrorPrefix + "compute pass '" + pass.Name + "': " + error); + return false; + } + + CommandBuffer commandBuffer = Commands; + Vk api = _context.Api; + + // No rendering scope encloses a dispatch, and a clear promoted into one of the + // pass's images lands before the pass reads or writes it. + _targets.EndRendering(commandBuffer); + foreach (Graph.ComputeBinding binding in pass.Bindings) + { + _targets.FlushPendingClears(commandBuffer, _textures.Get(binding.TextureId)!); + } + + _graph.OpenComputePass(Graph.ComputePassPlanner.Signature(_graph.NameId("compute:" + pass.Name), pass, + ComputeImageInfoOf)); + + foreach (Graph.ComputeBinding binding in pass.Bindings) + { + _textures.Require(_barriers, commandBuffer, _textures.Get(binding.TextureId)!, binding.BaseMip, + binding.MipCount, Graph.ComputePassPlanner.UsageOf(binding.Access)); + } + _barriers.Flush(commandBuffer); + + Span writes = pass.Bindings.Length <= 16 + ? stackalloc ComputeImageWrite[pass.Bindings.Length] + : new ComputeImageWrite[pass.Bindings.Length]; + for (int i = 0; i < pass.Bindings.Length; i++) + { + Graph.ComputeBinding binding = pass.Bindings[i]; + VulkanTexture texture = _textures.Get(binding.TextureId)!; + bool storage = Graph.ComputePassPlanner.IsStorage(binding.Access); + writes[i] = new ComputeImageWrite(binding.Binding, + storage ? DescriptorType.StorageImage : DescriptorType.CombinedImageSampler, + texture.ViewOfMips(binding.BaseMip, binding.MipCount), + storage ? default : _textures.Samplers.Get(binding.Linear ? ComputeLinear : ComputeNearest), + storage ? ImageLayout.General : ImageLayout.ShaderReadOnlyOptimal); + } + DescriptorSet set = _computeArenas[_frames.Current.Index].Get(program!.SetLayout, writes); + Pipeline pipeline = _compute.PipelineFor(program, pass.Specialization); + + api.CmdBindPipeline(commandBuffer, PipelineBindPoint.Compute, pipeline); + api.CmdBindDescriptorSets(commandBuffer, PipelineBindPoint.Compute, program.Layout, ComputeProgram.PassSet, 1, + &set, 0, null); + + foreach (Graph.ComputeDispatch dispatch in pass.Dispatches) + { + if (dispatch.PushConstants is { Length: > 0 } push) + { + fixed (byte* data = push) + { + api.CmdPushConstants(commandBuffer, program.Layout, ShaderStageFlags.ComputeBit, 0, (uint)push.Length, + data); + } + } + (uint x, uint y, uint z) = Graph.ComputePassPlanner.Groups(dispatch, pass, ComputeImageInfoOf, + program.LocalSizeX, program.LocalSizeY); + if (x == 0 || y == 0 || z == 0) continue; + api.CmdDispatch(commandBuffer, x, y, z); + _graph.NoteDispatch(); + if (RenderTrace.Enabled) + { + RenderTrace.Write("dispatch '" + pass.Name + "' program " + program.Id + " '" + program.Name + "' groups=" + + x + "x" + y + "x" + z); + } + } + return true; + } + + /// The mesh store. Tests only. + internal MeshManager MeshesForTests => _meshes; + + /// The context (and its allocator). Tests only. + internal VulkanContext ContextForTests => _context; + + /// The texture manager. Tests only. + internal TextureManager TexturesForTests => _textures; + + /// Where per-second backend counters go, when asked for. + private static readonly string? StatsLogPath = Environment.GetEnvironmentVariable("OPTIMUM_VULKAN_STATS"); + + /// + /// Leaves a marker the driver reports back if the GPU stops. Free when the + /// extension is absent; one small command otherwise. + /// + private void Checkpoint(CommandBuffer commandBuffer, nint marker) + { + if (_context.CheckpointsAvailable) _context.CmdSetCheckpoint(commandBuffer, marker); + } + + /// + /// What the GPU was doing when it was lost, from the driver's checkpoint and + /// fault records. VulkanResult.Check calls this on the first loss. + /// + /// Reading checkpoints wants the queue synchronised like any other queue + /// call, but the thread that noticed the loss may already hold the lock, or + /// another may be inside a submit that is about to fail. A bounded wait keeps + /// the crash report from deadlocking behind the crash it is describing. + /// + private string? DescribeDeviceLoss() + { + if (_context == null) return null; + + var text = new System.Text.StringBuilder(); + + if (_context.CheckpointsAvailable) + { + bool locked = System.Threading.Monitor.TryEnter(_context.QueueLock, 2000); + try + { + List<(PipelineStageFlags Stage, nint Marker)> checkpoints = _context.ReadQueueCheckpoints(); + if (checkpoints.Count == 0) + { + text.Append("The driver recorded no GPU checkpoints."); + } + else + { + text.Append("Last GPU checkpoint per stage -"); + foreach ((PipelineStageFlags stage, nint marker) in checkpoints) + { + text.Append(' ').Append(StageName(stage)).Append(": ") + .Append(CheckpointMarker.Describe(marker, ProgramNameOf)).Append(';'); + } + } + } + finally + { + if (locked) System.Threading.Monitor.Exit(_context.QueueLock); + } + } + else + { + text.Append("GPU checkpoints are not available on this driver."); + } + + string? fault = _context.DeviceFaultAvailable ? _context.ReadDeviceFault() : null; + if (fault != null) text.Append(' ').Append(fault).Append('.'); + + return text.ToString(); + } + + private string? ProgramNameOf(int programId) => + _programNames.TryGetValue(programId, out string? name) ? name : null; + + private static string StageName(PipelineStageFlags stage) => stage switch + { + PipelineStageFlags.TopOfPipeBit => "last started", + PipelineStageFlags.BottomOfPipeBit => "last completed", + _ => stage.ToString(), + }; + + /// + /// Ends the frame in two submissions. Submit A carries the upload batch and + /// the frame and signals the Frame timeline; only then does the CPU block on + /// vkAcquireNextImageKHR, with the whole frame already in flight. Submit B + /// (the present path: the flipped blit) waits on the frame at + /// COLOR_ATTACHMENT_OUTPUT and on the acquire semaphore at the image's first + /// use, and signals the image's present semaphore; then the image is presented. + /// + public void Present() + { + if (!_frameActive) return; + + TextureDump.NoteFrame(); + if (TextureDump.Wanted) DumpRequestedTextures(); + + // Clears no pass consumed land now: the image keeps them into the next frame. + _targets.FlushAllPendingClears(_frames.Current.CommandBuffer); + _targets.EndPass(_frames.Current.CommandBuffer); + if (_graph.Enabled) _graph.EndFrame(); + + // Any open rendering scope has to close before the command buffer ends. + _targets.EndRendering(_frames.Current.CommandBuffer); + + long presentEntry = System.Diagnostics.Stopwatch.GetTimestamp(); + // A slot first resolved while recording is written before its draws are submitted. + _bindless?.Flush(); + ulong renderValue = _frames.EndFrame(); + _frameActive = false; + // Seam S4: the frame's work is queued (Submit A). Stamped before the acquire, + // which is where the CPU may block, so the render-submit interval is recording + // time and nothing else. + Latency.Marker(_latencyFrameId, LatencyMarker.RenderSubmitEnd); + long frameSubmitted = System.Diagnostics.Stopwatch.GetTimestamp(); + + // Headless: nothing to present; the frame is submitted all the same. + if (_swapchain == null || _presentPath == null) return; + + bool acquired = _swapchain.TryAcquire(out PresentTarget target); + long acquireReturned = System.Diagnostics.Stopwatch.GetTimestamp(); + bool renderCompletedAtAcquire = _frames.Timeline.FrameCompleted >= renderValue; + ReportRebuildFailure(); + if (!acquired) + { + LastPresentTimingsForTests = new PresentTimings(presentEntry, frameSubmitted, acquireReturned, 0, + renderValue, 0, renderCompletedAtAcquire, false); + return; + } + + CommandBuffer presentCommands = _frames.BeginPresentCommands(); + Checkpoint(presentCommands, CheckpointMarker.PresentBlit(target.ImageIndex, _frameCounter)); + _presentPath.Record(presentCommands, target); + ulong presentValue = _frames.SubmitPresent( + target.AcquireSemaphore, _presentPath.AcquireWaitStage, renderValue, target.PresentSemaphore); + _swapchain.NotePresentSubmitted(target, presentValue); + long presentSubmitted = System.Diagnostics.Stopwatch.GetTimestamp(); + + // Seam S4: PresentStart and PresentEnd bracket vkQueuePresentKHR itself, and the + // present id the call was given closes the frame's report. + Latency.Marker(_latencyFrameId, LatencyMarker.PresentStart); + ulong presentId = _swapchain.Present(target, _latencyFrameId); + Latency.Marker(_latencyFrameId, LatencyMarker.PresentEnd); + Latency.OnPresent(_latencyFrameId, presentId); + LastPresentTimingsForTests = new PresentTimings(presentEntry, frameSubmitted, acquireReturned, presentSubmitted, + renderValue, presentValue, renderCompletedAtAcquire, true); + + long presentReturn = System.Diagnostics.Stopwatch.GetTimestamp(); + if (_lastPresentReturn != 0 && _vsync && + _missedVsyncs.NoteInterval((presentReturn - _lastPresentReturn) * 1000.0 / System.Diagnostics.Stopwatch.Frequency) && + _swapchain.PromoteToRelaxedFifo()) + { + MirrorValidationMessage("--- sustained missed vsyncs: swapchain promoted to FIFO_RELAXED"); + } + _lastPresentReturn = presentReturn; + } + + /// Stopwatch timestamps of one Present, for PresentDecouplingTests. + internal readonly record struct PresentTimings( + long PresentEntry, long FrameSubmitted, long AcquireReturned, long PresentSubmitted, + ulong RenderValue, ulong PresentValue, bool RenderCompletedAtAcquire, bool Presented); + + /// The last Present's timings. Tests only. + internal PresentTimings LastPresentTimingsForTests { get; private set; } + + /// The swapchain, null when headless. Tests only. + internal Swapchain? SwapchainForTests => _swapchain; + + private void ReportRebuildFailure() + { + string? failure = _swapchain?.RebuildFailure; + if (failure != null && failure != _reportedRebuildFailure) + { + AddDiagnostic("swapchain recreation failed: " + failure); + } + _reportedRebuildFailure = failure; + } + + /// + /// A new window size: the default framebuffer is rebuilt now (its old images + /// retire on the timelines), the swapchain at the next acquire. Nothing waits. + /// + public void Resize(int width, int height) + { + if (_swapchain == null || width <= 0 || height <= 0) return; + if ((uint)width == _windowWidth && (uint)height == _windowHeight) return; + + DestroyDefaultFramebuffer(); + CreateDefaultFramebuffer((uint)width, (uint)height); + _swapchain.RequestRebuild(_windowWidth, _windowHeight, _vsync); + } + + public void SetVSync(bool enabled) + { + if (_vsync == enabled) return; + _vsync = enabled; + _missedVsyncs.Reset(); + _swapchain?.RequestRebuild(_windowWidth, _windowHeight, _vsync); + } + + private CommandBuffer Commands => _frames.Current.CommandBuffer; + + // ------------------------------------------------------------------- teardown + + /// + /// Writes the driver's pipeline cache and the pipeline-key log for the next launch, after + /// the device is idle; opportunistic saves during the session come from + /// . A failed write costs the next launch its + /// warm start, nothing more. + /// + private void SavePipelineCache() + { + if (_pipelinePersistence == null || _pipelines == null) return; + _pipelinePersistence.SaveAtShutdown(_pipelines); + } + + public void Dispose() + { + if (_disposed) return; + _disposed = true; + + if (_context != null) + { + VulkanStats.WaitDeviceIdle(_context.Api, _context.Device); + } + + // Background compiles read program modules and layouts, and a background save + // reads the driver cache: both end before anything they use is destroyed. + _pipelines?.StopBackgroundCompiles(); + _pipelinePersistence?.WaitForPendingSave(); + + foreach (ShaderProgramResources program in _programs.Values) program.Dispose(); + _programs.Clear(); + _compute?.Dispose(); + // The shared pipeline layout before the table's set layout it names; the + // table's placeholders are textures and go with the texture manager. + _sharedLayout?.Dispose(); + _bindless?.Dispose(); + + _uniformBuffers.Clear(); + + _queryRing?.Dispose(); + _readbacks?.Dispose(); + + foreach (VulkanBuffer? indirect in _indirectBuffers) indirect?.Dispose(); + foreach (VulkanBuffer overflow in _indirectOverflow) overflow.Dispose(); + _indirectOverflow.Clear(); + foreach (DescriptorArena arena in _descriptorArenas) arena.Dispose(); + foreach (ComputeDescriptorArena arena in _computeArenas) arena.Dispose(); + _defaultAttributes?.Dispose(); + _placeholderUniforms?.Dispose(); + _swapchain?.Dispose(); + _shaderCompiler?.Dispose(); + _frames?.Dispose(); + _descriptors?.Dispose(); + SavePipelineCache(); + _pipelines?.Dispose(); + _targets?.Dispose(); + _meshes?.Dispose(); + _textures?.Dispose(); + if (_context != null && ReferenceEquals(VulkanStats.MemorySource, _context.Allocator)) + { + VulkanStats.MemorySource = null; + } + DisposeLatency(); + _context?.Dispose(); + } +} diff --git a/Optimum.Tests/ItemRenderInfoReuseTests.cs b/Optimum.Tests/ItemRenderInfoReuseTests.cs index 95f46a26..c50e0460 100644 --- a/Optimum.Tests/ItemRenderInfoReuseTests.cs +++ b/Optimum.Tests/ItemRenderInfoReuseTests.cs @@ -21,6 +21,7 @@ namespace Optimum.Tests; /// call it and may retain the result), only the internal per-slot path reuses. /// Plus the config round-trip and the patch/patcher wiring. /// +[Collection("OptimumConfig")] public class ItemRenderInfoReuseTests { private static string RepoRoot() @@ -109,6 +110,7 @@ public void ItemRenderInfoReuse_ConfigRoundTrips() finally { OptimumConfig.ItemRenderInfoReuseEnabled = true; + OptimumConfig.SetDataPath(null); try { Directory.Delete(dir, true); } catch { } } } diff --git a/Optimum.Tests/fsr-pipeline-coverage-tests.cs b/Optimum.Tests/fsr-pipeline-coverage-tests.cs index 3d9c2956..a74c2594 100644 --- a/Optimum.Tests/fsr-pipeline-coverage-tests.cs +++ b/Optimum.Tests/fsr-pipeline-coverage-tests.cs @@ -72,15 +72,23 @@ public void TerrainBiasCoversTextureObjectsAndCustomSamplers() string shaderRegistry = ReadPatchedOrSource( "patches/VintagestoryLib/Vintagestory.Client.NoObf/ShaderRegistry.cs.patch", "build/VintagestoryLib/Vintagestory.Client.NoObf/ShaderRegistry.cs"); - - // Bias must be skipped entirely at native res (RenderScale >= 1.0) so - // rendering matches vanilla exactly - vanilla never sets these - // TexParameter/SamplerParameter calls at all. - Assert.Contains("if (ClientSettings.OptimumRenderScale >= 1.0f)", chunkRenderer); - Assert.Contains("MathF.Log2(Math.Clamp(Vintagestory.API.Config.OptimumConfig.EffectiveRenderScale, 0.5f, 1.0f))", chunkRenderer); - Assert.Contains("(TextureParameterName)34049, textureLodBias", chunkRenderer); - Assert.Contains("if (OptimumConfig.EffectiveRenderScale < 1.0f)", shaderRegistry); - Assert.Contains("(SamplerParameterName)34049, terrainLodBias", shaderRegistry); + string platform = ReadPatchedOrSource( + "patches/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs.patch", + "build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string config = File.ReadAllText(PatchReader.FindRepositoryFile("sources/VintagestoryApi/Config/OptimumConfig.cs")); + + // One effective value drives both texture and sampler parameters, including + // the zero-bias path at native scale with TAA disabled. + Assert.Contains("if (scale < 1.0f)", config); + Assert.Contains("MathF.Log2(Math.Clamp(scale, 0.5f, 1.0f))", config); + Assert.Contains("float textureLodBias = Vintagestory.API.Config.OptimumConfig.EffectiveTerrainLodBias", chunkRenderer); + Assert.Contains("if (textureLodBias == 0f)", chunkRenderer); + Assert.Contains("game.Platform.SetTextureLodBias(textureIds, bias)", chunkRenderer); + Assert.Contains("float terrainLodBias = OptimumConfig.EffectiveTerrainLodBias", shaderRegistry); + Assert.Contains("if (terrainLodBias != 0f)", shaderRegistry); + Assert.Contains("platform.SetSamplerLodBias(sampler, bias)", shaderRegistry); + Assert.Contains("(TextureParameterName)34049, bias", platform); + Assert.Contains("(SamplerParameterName)34049, bias", platform); Assert.Contains("terrainTexLinear", shaderRegistry); } diff --git a/Optimum.Tests/oit-framebuffer-rebuild-coverage-tests.cs b/Optimum.Tests/oit-framebuffer-rebuild-coverage-tests.cs index 5b62cd47..c1270e6a 100644 --- a/Optimum.Tests/oit-framebuffer-rebuild-coverage-tests.cs +++ b/Optimum.Tests/oit-framebuffer-rebuild-coverage-tests.cs @@ -43,7 +43,8 @@ public void OitRebuildComparesPreviousFramebufferToCurrentBeforeReplacingIt() [Fact] public void OitRevealAttachesToColorAttachment0_NotOverwritingVanilla() { - string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/SystemRenderOITLayers.cs"); + string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string native = Read("Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs"); // OIT reveal texture attaches to ColorAttachment0 (36064). This is by design: // the oit.fsh shader writes to layout(location = 0) which IS ColorAttachment0. @@ -51,24 +52,30 @@ public void OitRevealAttachesToColorAttachment0_NotOverwritingVanilla() // from the FBO, but MergeTransparentRenderPass reads it by texture ID from // the shader uniform (not from attachment), so it reads the Optimum reveal data. Assert.Contains("(FramebufferAttachment)36064", source); + Assert.Contains("EnumFramebufferAttachment.ColorAttachment0, revealTexture", native); } [Fact] public void OitAccumulationLayersAttachToSlots3Through5() { - string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/SystemRenderOITLayers.cs"); + string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string native = Read("Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs"); // Accumulation layers attach to ColorAttachment3-5 (36067, 36068, 36069) // matching oit.fsh layout(location = 3/4/5). Assert.Contains("(FramebufferAttachment)36067", source); Assert.Contains("(FramebufferAttachment)36068", source); Assert.Contains("(FramebufferAttachment)36069", source); + Assert.Contains("EnumFramebufferAttachment.ColorAttachment3, accumTexture, 0", native); + Assert.Contains("EnumFramebufferAttachment.ColorAttachment4, accumTexture, 1", native); + Assert.Contains("(EnumFramebufferAttachment)36069, accumTexture, 2", native); } [Fact] public void OitDrawBuffersMatchesShaderOutputLocations() { - string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/SystemRenderOITLayers.cs"); + string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string native = Read("Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs"); // DrawBuffers must declare 6 attachments (0-5) matching oit.fsh outputs. // Using 6, not 7: attachment 6 would be unused by shaders. @@ -76,6 +83,7 @@ public void OitDrawBuffersMatchesShaderOutputLocations() Assert.Contains("DrawBuffersEnum.ColorAttachment0", source); Assert.Contains("DrawBuffersEnum.ColorAttachment5", source); Assert.DoesNotContain("DrawBuffersEnum.ColorAttachment6", source); + Assert.Contains("StateDrawBuffers(transparent.FboId, 0x3F)", native); } [Fact] @@ -90,23 +98,30 @@ public void OitDisablesFlagPreventsFurtherRenderCalls() [Fact] public void OitBlendFuncPreservesVanillaAttachments0And1() { - string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/SystemRenderOITLayers.cs"); + string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string native = Read("Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs"); // Attachments 0 and 1 use DST_COLOR * ZERO blend (774 = GL_DST_COLOR, 0 = GL_ZERO). // This multiplies existing content by incoming fragment, preserving reveal semantics. Assert.Contains("GL.BlendFunc(0, (BlendingFactorSrc)774, (BlendingFactorDest)0)", source); Assert.Contains("GL.BlendFunc(1, (BlendingFactorSrc)774, (BlendingFactorDest)0)", source); + Assert.Contains("StateSlotBlendFunc(0, 774, 0, 774, 0)", native); + Assert.Contains("StateSlotBlendFunc(1, 774, 0, 774, 0)", native); } [Fact] public void OitAccumulationBlendFuncUsesAdditiveBlend() { - string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/SystemRenderOITLayers.cs"); + string source = Read("build/VintagestoryLib/Vintagestory.Client.NoObf/ClientPlatformWindows.cs"); + string native = Read("Optimum.Render.Vulkan/Platform/VulkanClientPlatform.Leaf.cs"); // Attachments 3-5 use ONE + ONE additive blend (1 = GL_ONE). Assert.Contains("GL.BlendFunc(3, (BlendingFactorSrc)1, (BlendingFactorDest)1)", source); Assert.Contains("GL.BlendFunc(4, (BlendingFactorSrc)1, (BlendingFactorDest)1)", source); Assert.Contains("GL.BlendFunc(5, (BlendingFactorSrc)1, (BlendingFactorDest)1)", source); + Assert.Contains("StateSlotBlendFunc(3, 1, 1, 1, 1)", native); + Assert.Contains("StateSlotBlendFunc(4, 1, 1, 1, 1)", native); + Assert.Contains("StateSlotBlendFunc(5, 1, 1, 1, 1)", native); } private static string Read(string relativePath) diff --git a/Optimum.Tests/shader-compatibility-report-tests.cs b/Optimum.Tests/shader-compatibility-report-tests.cs index 8925275c..6b9a55d9 100644 --- a/Optimum.Tests/shader-compatibility-report-tests.cs +++ b/Optimum.Tests/shader-compatibility-report-tests.cs @@ -5,6 +5,7 @@ namespace Optimum.Tests; +[Collection("OptimumConfig")] public sealed class ShaderCompatibilityReportTests : IDisposable { private readonly string _tempDataDir; diff --git a/README.md b/README.md index 1322a0a6..10082c90 100644 --- a/README.md +++ b/README.md @@ -204,6 +204,24 @@ after editing it. When troubleshooting world-generation problems, the four relevant keys are `ChunkReadPoolEnabled`, `ChunkReadPoolWorkers`, `ChunkDeserializeParallel`, and `ChunkDeserializeParallelMinY`. +### Experimental Vulkan renderer + +The Vulkan backend is opt-in on this branch. Set `"Renderer": "vulkan"` in the +active data path's `ModConfig/optimum.json` and restart the client. The startup +log must contain `[Optimum] Vulkan renderer`; if Vulkan initialization fails, +the client may reopen with OpenGL. Set `"Renderer": "opengl"` to return to the +default backend. + +`"Taa": true` enables temporal antialiasing. With `"AmbientOcclusion": "auto"`, +Vulkan selects GTAO while TAA is active and uses the game's SSAO otherwise; +OpenGL continues to use the game's SSAO. The TAA sharpen runs after bloom, god +rays and final composition. FSR 1's RCAS takes its place when FSR is active. +GPU pass timings can be logged with `OPTIMUM_VULKAN_PASS_TIMES=1`; Vulkan +validation can be enabled with `OPTIMUM_VULKAN_VALIDATION=1` and +`OPTIMUM_VULKAN_VALIDATION_FEATURES=sync,best` when the validation layer is +installed. See [Vulkan acceptance](docs/vulkan.md) for the renderer +confirmation, parity and pacing procedures. + ## Build ### Targeting a Vintage Story version diff --git a/VintageStory.slnx b/VintageStory.slnx index 47d6300a..7dd4062a 100644 --- a/VintageStory.slnx +++ b/VintageStory.slnx @@ -18,9 +18,21 @@ + + + + + + + + + + Exe + net10.0 + Optimum.Shaders.Compiler + Optimum.Shaders.Compiler + annotations + true + + true + + + + + + + + + + ..\..\.vanilla\win-x64\vintagestory\VintagestoryAPI.dll + true + + + + + $([System.IO.Path]::GetFullPath('$(MSBuildThisFileDirectory)..\..\')) + $(OptimumRepositoryRoot)sources$([System.IO.Path]::DirectorySeparatorChar)shaders-vk + + $(OptimumRepositoryRoot)bin$([System.IO.Path]::DirectorySeparatorChar)$(Configuration)$([System.IO.Path]::DirectorySeparatorChar)net10.0 + $(NativeShaderOutputRoot)$([System.IO.Path]::DirectorySeparatorChar)shaders-vk$([System.IO.Path]::DirectorySeparatorChar)shaders.manifest.json + $(MSBuildProjectDirectory)/$(BaseIntermediateOutputPath)shaders-vk.inputs + $(MSBuildProjectDirectory)/$(BaseIntermediateOutputPath)shaders-vk.stamp + $(DOTNET_HOST_PATH) + dotnet + + + + + + + + + + + + + + + + + + + diff --git a/tools/shader-compiler/Program.cs b/tools/shader-compiler/Program.cs new file mode 100644 index 00000000..3c60ce7e --- /dev/null +++ b/tools/shader-compiler/Program.cs @@ -0,0 +1,9 @@ +using System; +using Optimum.Render.Vulkan.Shaders; + +namespace Optimum.Shaders.Compiler; + +internal static class Program +{ + private static int Main(string[] args) => NativeShaderTool.Run(args, Console.Out, Console.Error); +}