diff --git a/README.md b/README.md index 9803b1a..233457c 100644 --- a/README.md +++ b/README.md @@ -18,7 +18,7 @@ Press `U` (sometimes `O`) or, use the mod menu config to control the render scal ## Other features -On 1.21.11 and above, there is experimental support for **FSR 1.0**! No plans to support DLSS or FSR 2.0+. +**FSR 1** support! No plans to support DLSS or FSR 2+. Also, there is experimental support for **Dynamic Scaling** (1.4.0-alpha.6+), which tries to keep the game at your targeted FPS by changing the multiplier while you play. diff --git a/src/main/java/dev/zelo/renderscale/RenderScale.java b/src/main/java/dev/zelo/renderscale/RenderScale.java index 345257d..95658f4 100644 --- a/src/main/java/dev/zelo/renderscale/RenderScale.java +++ b/src/main/java/dev/zelo/renderscale/RenderScale.java @@ -96,20 +96,96 @@ public class RenderScale { private RenderTarget intermediateTarget; //? >= 1.21.11 { - public static RenderPipeline FSR_EASU_PIPELINE = buildFsrEasuPipeline(); + public static final RenderPipeline FSR_EASU_PIPELINE = createFsrPipeline("easu", 0); + public static final RenderPipeline FSR_RCAS_PIPELINE = createFsrPipeline("rcas", 0); - private static RenderPipeline buildFsrEasuPipeline() { - RenderPipeline.Builder builder = fullscreenBuilder("pipeline/fsr_easu", "core/easu"); + private Object fsrDevice; + private RenderPipeline fsrEasuPipeline = FSR_EASU_PIPELINE; + private RenderPipeline fsrRcasPipeline = FSR_RCAS_PIPELINE; + + private static RenderPipeline createFsrPipeline(String pass, int fp16Extension) { + String suffix = fp16Extension == 0 ? "" : "_fp16"; + RenderPipeline.Builder builder = fullscreenBuilder("pipeline/fsr_" + pass + suffix, "core/" + pass + suffix); // The explicit-gather workaround only applies to the RenderPearl // pipeline API; older families keep the native textureGather path. //? >=26.3 { - builder = builder.withShaderDefine("RENDERSCALE_EXPLICIT_GATHER"); + if ("easu".equals(pass)) { + builder.withShaderDefine("RENDERSCALE_EXPLICIT_GATHER"); + } //?} + if (fp16Extension != 0) { + builder.withShaderDefine("RENDERSCALE_FP16", fp16Extension); + } return builder.build(); } - public static RenderPipeline FSR_RCAS_PIPELINE = - fullscreenPipeline("pipeline/fsr_rcas", "core/rcas"); + private void selectFsrPipelines() { + var device = RenderSystem.getDevice(); + if (fsrDevice == device) return; + fsrDevice = device; + fsrEasuPipeline = FSR_EASU_PIPELINE; + fsrRcasPipeline = FSR_RCAS_PIPELINE; + + int extension = 0; + //? >=26.2 { + String backend = device.getDeviceInfo().backendName(); + //?} else + //String backend = device.getBackendName(); + if ("OpenGL".equals(backend)) { + // DeviceInfo lists only extensions used by vanilla, not all supported + // extensions. Query the active context after confirming the backend. + var capabilities = org.lwjgl.opengl.GL.getCapabilities(); + // NVIDIA exposes FP16 (and accepts our GLSL 450 shaders) even in + // Minecraft's OpenGL 3.3 context. Check the shader extensions rather + // than the context version, then validate both compiled pipelines. + if (capabilities.GL_AMD_gpu_shader_half_float) extension = 2; + else if (capabilities.GL_NV_gpu_shader5) extension = 3; + else { + int count = org.lwjgl.opengl.GL30C.glGetInteger(org.lwjgl.opengl.GL30C.GL_NUM_EXTENSIONS); + for (int i = 0; i < count; i++) { + String name = org.lwjgl.opengl.GL30C.glGetStringi(org.lwjgl.opengl.GL30C.GL_EXTENSIONS, i); + if ("GL_EXT_shader_explicit_arithmetic_types_float16".equals(name) + || "GL_EXT_shader_explicit_arithmetic_types".equals(name)) { + extension = 1; + break; + } + } + } + } + //? >=26.2 { + if ("Vulkan".equals(backend) && dev.zelo.renderscale.compat.vulkan.VulkanFsrSupport.isEnabled(device)) { + extension = 1; + } + //?} + if (extension == 0) { + Constants.LOG.info("FSR1: using FP32 (shader FP16 is unavailable)"); + return; + } + //? >=26.3 { + // ShaderC consumes EXT syntax; SPIRV-Cross emits the driver's extension. + extension = 1; + //?} + var easu = createFsrPipeline("easu", extension); + var rcas = createFsrPipeline("rcas", extension); + try { + //? >=26.3 { + boolean valid = RenderSystem.getCompiledPipelineNullable(easu) != null + && RenderSystem.getCompiledPipelineNullable(rcas) != null; + //?} else { + /*boolean valid = device.precompilePipeline(easu).isValid() + && device.precompilePipeline(rcas).isValid(); + *///?} + if (valid) { + fsrEasuPipeline = easu; + fsrRcasPipeline = rcas; + Constants.LOG.info("FSR1: using FP16"); + } else { + Constants.LOG.warn("FSR1: FP16 shader compilation failed; using FP32"); + } + } catch (RuntimeException exception) { + Constants.LOG.warn("FSR1: FP16 shader compilation failed; using FP32", exception); + } + } public static RenderPipeline RGSS_PIPELINE = fullscreenPipeline("pipeline/rgss", "core/rgss"); @@ -377,6 +453,26 @@ public double getRenderScaleFactor() { : getConfig().getScale(); } + public String getScalingMode() { + double scale = getRenderScaleFactor(); + boolean upsampling = scale < 1.0; + boolean downsampling = scale > 1.0; + if (renderTarget != null && clientRenderTarget != null) { + upsampling = renderTarget.width < clientRenderTarget.width + || renderTarget.height < clientRenderTarget.height; + downsampling = renderTarget.width > clientRenderTarget.width + || renderTarget.height > clientRenderTarget.height; + } + String direction = downsampling ? "Downsampling" : upsampling ? "Upsampling" : "Native"; + // Match the blit pass: FSR also runs at native resolution, but not when downsampling. + //? >= 1.21.11 { + if (getConfig().fsr && !downsampling) { + return direction + ", FSR1 " + (fsrEasuPipeline == FSR_EASU_PIPELINE ? "FP32" : "FP16"); + } + //?} + return direction + (getConfig().getFilter() ? ", Linear" : ", Nearest"); + } + public void updateDynamicScale() { RenderScaleConfig config = getConfig(); if (dynamicScaleLevel != client.level) { @@ -491,9 +587,10 @@ public void blitAndBlendToTexture(final RenderTarget input, final RenderTarget o /*try (RenderPass renderPass = RenderSystem.getDevice().createCommandEncoder().createRenderPass(() -> "Blit render target", output.getColorTextureView(), OptionalInt.empty())) { *///? } else if (getConfig().fsr && input.width <= output.width && input.height <= output.height) { + selectFsrPipelines(); RenderTarget intermediate = ensureIntermediateTarget(output.width, output.height); - fullscreenPass("FSR: EASU", FSR_EASU_PIPELINE, intermediate, input, filter); - fullscreenPass("FSR: RCAS", FSR_RCAS_PIPELINE, output, intermediate, filter); + fullscreenPass("FSR: EASU", fsrEasuPipeline, intermediate, input, filter); + fullscreenPass("FSR: RCAS", fsrRcasPipeline, output, intermediate, filter); } else if (input.width > output.width && input.height > output.height && getConfig().getDownscaleFilter() != RenderScaleConfig.DownscaleFilter.BILINEAR) { // Supersampling: the render target is larger than the output, so downscale diff --git a/src/main/java/dev/zelo/renderscale/compat/vulkan/VulkanFsrSupport.java b/src/main/java/dev/zelo/renderscale/compat/vulkan/VulkanFsrSupport.java new file mode 100644 index 0000000..29423fa --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/compat/vulkan/VulkanFsrSupport.java @@ -0,0 +1,54 @@ +//? >=26.2 { +package dev.zelo.renderscale.compat.vulkan; + +//? >=26.3 { +import com.mojang.renderpearl.api.device.GpuDevice; +import com.mojang.renderpearl.backend.vulkan.VulkanDevice; +import com.mojang.renderpearl.backend.vulkan.VulkanFeatureSets; +import com.mojang.renderpearl.backend.vulkan.init.FeatureSet; +import com.mojang.renderpearl.backend.vulkan.init.VulkanFeature; +//?} else { +/*import com.mojang.blaze3d.systems.GpuDevice; +import com.mojang.blaze3d.vulkan.VulkanDevice; +import com.mojang.blaze3d.vulkan.VulkanBackend; +import com.mojang.blaze3d.vulkan.init.VulkanFeature; +*///?} +import dev.zelo.renderscale.mixin.accessors.MixinGpuDeviceAccessor; +import org.lwjgl.vulkan.VkDevice; + +import java.util.Collections; +import java.util.Map; +import java.util.Set; +import java.util.WeakHashMap; + +public final class VulkanFsrSupport { + // Both Vulkan backends require Vulkan 1.2. Reuse their existing feature + // structure; no KHR extension or additional 16-bit storage feature is needed. + public static final VulkanFeature SHADER_FLOAT16 = new VulkanFeature( + //? >=26.3 { + VulkanFeatureSets.VK12_FEATURES_STRUCT, + //?} else + //VulkanBackend.VK12_FEATURES_STRUCT, + "shaderFloat16", org.lwjgl.vulkan.VkPhysicalDeviceVulkan12Features.SHADERFLOAT16); + + //? >=26.3 { + public static final FeatureSet FP16 = new FeatureSet("RenderScale FSR1 FP16", Set.of(), Set.of(SHADER_FLOAT16)); + //?} + + // Track successful logical-device creation, not just physical-device support. + // Weak keys avoid retaining devices after a backend switch or failed startup. + private static final Map ENABLED = Collections.synchronizedMap(new WeakHashMap<>()); + + private VulkanFsrSupport() {} + + public static void recordEnabled(VkDevice device, boolean enabled) { + ENABLED.put(device, enabled); + } + + public static boolean isEnabled(GpuDevice device) { + return device instanceof MixinGpuDeviceAccessor accessor + && accessor.renderScale$getBackend() instanceof VulkanDevice vulkan + && Boolean.TRUE.equals(ENABLED.get(vulkan.vkDevice())); + } +} +//?} diff --git a/src/main/java/dev/zelo/renderscale/gametest/ForgeShaderCompatibilityTest.java b/src/main/java/dev/zelo/renderscale/gametest/ForgeShaderCompatibilityTest.java new file mode 100644 index 0000000..09376de --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/gametest/ForgeShaderCompatibilityTest.java @@ -0,0 +1,161 @@ +//? 1.20.1 && forge { +/*package dev.zelo.renderscale.gametest; + +import com.mojang.blaze3d.pipeline.RenderTarget; +import com.mojang.blaze3d.pipeline.TextureTarget; +import dev.zelo.renderscale.Constants; +import dev.zelo.renderscale.RenderScale; +import net.irisshaders.iris.Iris; +import net.irisshaders.iris.gl.framebuffer.GlFramebuffer; +import net.irisshaders.iris.pipeline.IrisRenderingPipeline; +import net.irisshaders.iris.targets.Blaze3dRenderTargetExt; +import net.irisshaders.iris.targets.RenderTargets; +import net.minecraft.client.Minecraft; +import net.minecraft.client.gui.screens.PauseScreen; +import net.minecraft.client.gui.screens.TitleScreen; +import net.minecraftforge.event.TickEvent; +import org.lwjgl.opengl.GL11C; +import org.lwjgl.opengl.GL30C; +import org.lwjgl.opengl.GL45C; + +import java.lang.reflect.Field; +import java.nio.file.Files; + +// Run in a disposable game directory with Oculus, a shaderpack, and +// -Drenderscale.shaderCompatibilityTest=true. Render ticks continue while paused. +public final class ForgeShaderCompatibilityTest { + private static int step; + private static int frames; + private static long started; + private static boolean done; + + private ForgeShaderCompatibilityTest() {} + + public static void frame(TickEvent.RenderTickEvent event) { + if (event.phase != TickEvent.Phase.END || done) return; + Minecraft client = Minecraft.getInstance(); + if (started == 0) started = System.nanoTime(); + try { + if (System.nanoTime() - started > 180_000_000_000L) throw new AssertionError("Timed out at step " + step); + if (step == 0) { + if (!(client.screen instanceof TitleScreen) || client.getOverlay() != null) return; + client.options.pauseOnLostFocus = false; + // Reproduce modpacks that request stencil before entering a world. + client.getMainRenderTarget().enableStencil(); + setScale(0.25f); + RenderScaleAutoTest.INSTANCE.createWorld(client); + step++; + return; + } + if (client.level == null || client.player == null) return; + if (step == 1 && client.screen != null) return; + if (++frames < 30) return; + frames = 0; + verifyTargets(); + Constants.LOG.info("Shader compatibility test step {} passed (scale={}, paused={})", + step, RenderScale.getConfig().scale, client.isPaused()); + switch (step++) { + case 1 -> setScale(1.0f); + case 2 -> { + // Force the version collision instead of relying on incidental + // window resizes or a particular driver's texture-ID allocation. + RenderScale scale = RenderScale.getInstance(); + while (version(scale.clientRenderTarget) < version(scale.renderTarget)) { + resize(scale.clientRenderTarget); + } + while (version(scale.renderTarget) < version(scale.clientRenderTarget)) { + resize(scale.renderTarget); + } + Iris.reload(); // Pipeline is created against the native target. + client.getMainRenderTarget().bindWrite(true); + } + case 3 -> client.setScreen(new PauseScreen(true)); + case 4 -> { + require(client.isPaused(), "Pause screen did not pause the game"); + client.setScreen(null); + } + case 5 -> setScale(0.25f); + case 6 -> client.setScreen(new PauseScreen(true)); + case 7 -> { + require(client.isPaused(), "Pause screen did not pause the game"); + client.setScreen(null); + } + case 8 -> finish(client, null); + default -> throw new AssertionError("Unexpected step"); + } + } catch (Throwable error) { + finish(client, error); + } + } + + private static void setScale(float value) { + RenderScale.getConfig().scale = value; + RenderScale.getConfig().targetFrameRate = 0; + RenderScale.CONFIG.save(); + } + + private static int version(RenderTarget target) { + return ((Blaze3dRenderTargetExt) target).iris$getDepthBufferVersion(); + } + + private static void resize(RenderTarget target) { + target.resize(target.width, target.height, Minecraft.ON_OSX); + } + + private static void verifyTargets() throws ReflectiveOperationException { + RenderScale scale = RenderScale.getInstance(); + require(scale.renderTarget != null && scale.clientRenderTarget != null, "Missing targets"); + require(scale.renderTarget.isStencilEnabled() && scale.clientRenderTarget.isStencilEnabled(), "Stencil lost during target swap"); + require(Minecraft.getInstance().getMainRenderTarget() == scale.clientRenderTarget, "Main target not restored for UI"); + Object pipeline = Iris.getPipelineManager().getPipelineNullable(); + require(pipeline instanceof IrisRenderingPipeline, "Shader pipeline not active"); + RenderTargets targets = (RenderTargets) field(pipeline, "renderTargets"); + int depth = scale.renderTarget.getDepthTextureId(); + require(targets.getDepthTexture() == depth, "Oculus retained another target's depth texture"); + GlFramebuffer source = (GlFramebuffer) field(targets, "depthSourceFb"); + int attachment = GL45C.glGetNamedFramebufferAttachmentParameteri(source.getId(), GL30C.GL_DEPTH_ATTACHMENT, + GL30C.GL_FRAMEBUFFER_ATTACHMENT_OBJECT_NAME); + int stencil = GL45C.glGetNamedFramebufferAttachmentParameteri(source.getId(), GL30C.GL_STENCIL_ATTACHMENT, + GL30C.GL_FRAMEBUFFER_ATTACHMENT_OBJECT_NAME); + require(attachment == depth && stencil == depth, "Oculus depth and stencil are not the scaled texture"); + for (GlFramebuffer framebuffer : (Iterable) field(targets, "ownedFramebuffers")) { + require(GL45C.glCheckNamedFramebufferStatus(framebuffer.getId(), GL30C.GL_FRAMEBUFFER) + == GL30C.GL_FRAMEBUFFER_COMPLETE, "Incomplete Oculus framebuffer " + framebuffer.getId()); + } + // Exercise the full-size depth copy used by post-processing/particle mods. + TextureTarget copy = new TextureTarget(scale.clientRenderTarget.width, scale.clientRenderTarget.height, true, Minecraft.ON_OSX); + try { + copy.enableStencil(); + copy.copyDepthFrom(scale.renderTarget); + require(GL11C.glGetError() == GL11C.GL_NO_ERROR, "OpenGL error during rendering or depth copy"); + } finally { + copy.destroyBuffers(); + scale.clientRenderTarget.bindWrite(true); + } + FramebufferPresentationTest.verify(); + } + + private static Object field(Object owner, String name) throws ReflectiveOperationException { + Field field = owner.getClass().getDeclaredField(name); + field.setAccessible(true); + return field.get(owner); + } + + private static void require(boolean condition, String message) { + if (!condition) throw new AssertionError(message); + } + + private static void finish(Minecraft client, Throwable error) { + done = true; + try { + Files.writeString(client.gameDirectory.toPath().resolve("shader-compatibility-result.txt"), + error == null ? "PASS\n" : "FAIL: " + error + "\n"); + } catch (java.io.IOException io) { + Constants.LOG.error("Could not write shader compatibility result", io); + } + if (error == null) Constants.LOG.info("Shader compatibility test PASSED"); + else Constants.LOG.error("Shader compatibility test FAILED at step " + step, error); + client.stop(); + } +} +*///?} diff --git a/src/main/java/dev/zelo/renderscale/gametest/FramebufferPresentationTest.java b/src/main/java/dev/zelo/renderscale/gametest/FramebufferPresentationTest.java new file mode 100644 index 0000000..e783148 --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/gametest/FramebufferPresentationTest.java @@ -0,0 +1,89 @@ +//? 1.20.1 && forge { +/*package dev.zelo.renderscale.gametest; + +import com.mojang.blaze3d.pipeline.TextureTarget; +import com.mojang.blaze3d.platform.GlStateManager; +import com.mojang.blaze3d.platform.NativeImage; +import com.mojang.blaze3d.systems.RenderSystem; +import dev.zelo.renderscale.Constants; +import net.minecraft.client.Minecraft; +import net.minecraft.client.Screenshot; +import org.lwjgl.opengl.GL11C; +import org.lwjgl.opengl.GL30C; +import org.lwjgl.system.MemoryStack; + +import java.nio.ByteBuffer; + +// F2 reads a texture, so screenshot-only tests cannot detect stale window pixels. +final class FramebufferPresentationTest { + private FramebufferPresentationTest() {} + + static void verify() { + Minecraft client = Minecraft.getInstance(); + TextureTarget source = new TextureTarget(32, 32, false, Minecraft.ON_OSX); + TextureTarget destination = new TextureTarget(32, 32, false, Minecraft.ON_OSX); + try { + RenderSystem.disableScissor(); + RenderSystem.colorMask(true, true, true, true); + for (float alpha : new float[]{0.0f, 0.25f, 0.5f, 1.0f}) { + source.setClearColor(0.125f, 0.5f, 0.875f, alpha); + source.clear(Minecraft.ON_OSX); + int expected; + try (NativeImage screenshot = Screenshot.takeScreenshot(source)) { + expected = screenshot.getPixelRGBA(16, 16); + } + + // Prime vanilla's cached shader blend mode, then simulate a GUI + // leaving ordinary alpha blending enabled before presentation. + destination.bindWrite(true); + source.blitToScreen(32, 32, false); + GlStateManager._glBindFramebuffer(GL30C.GL_FRAMEBUFFER, 0); + GL11C.glDrawBuffer(GL11C.GL_BACK); + GL11C.glReadBuffer(GL11C.GL_BACK); + RenderSystem.colorMask(true, true, true, true); + RenderSystem.clearColor(1, 0, 1, 1); + RenderSystem.clear(GL11C.GL_COLOR_BUFFER_BIT, Minecraft.ON_OSX); + RenderSystem.enableBlend(); + RenderSystem.defaultBlendFunc(); + + // This is the overload Minecraft uses to present its main target. + source.blitToScreen(client.getWindow().getWidth(), client.getWindow().getHeight()); + requirePixel(expected, "Window retained old pixels at source alpha=" + alpha); + } + + // Explicit false is still used by RenderScale's world-to-UI copy. + // Preserve that caller's requested blending behavior. + source.setClearColor(0.125f, 0.5f, 0.875f, 0); + source.clear(Minecraft.ON_OSX); + destination.setClearColor(1, 0, 1, 1); + destination.clear(Minecraft.ON_OSX); + destination.bindWrite(true); + RenderSystem.enableBlend(); + RenderSystem.defaultBlendFunc(); + source.blitToScreen(32, 32, false); + requirePixel(0xFFFF00FF, "Explicit blended copy no longer preserves the destination"); + if (GL11C.glGetError() != GL11C.GL_NO_ERROR) throw new AssertionError("OpenGL error during presentation test"); + Constants.LOG.info("Framebuffer presentation passed: 4 source alpha values, F2/window RGB agreement, explicit blended copy"); + } finally { + source.destroyBuffers(); + destination.destroyBuffers(); + RenderSystem.disableBlend(); + client.getMainRenderTarget().bindWrite(true); + } + } + + private static void requirePixel(int expected, String message) { + try (MemoryStack stack = MemoryStack.stackPush()) { + ByteBuffer pixel = stack.malloc(4); + GL11C.glReadPixels(16, 16, 1, 1, GL11C.GL_RGBA, GL11C.GL_UNSIGNED_BYTE, pixel); + for (int channel = 0; channel < 3; channel++) { + int actual = Byte.toUnsignedInt(pixel.get(channel)); + int wanted = (expected >>> (channel * 8)) & 255; + if (Math.abs(actual - wanted) > 2) { + throw new AssertionError(message + ": RGB channel " + channel + " expected=" + wanted + " actual=" + actual); + } + } + } + } +} +*///?} diff --git a/src/main/java/dev/zelo/renderscale/gametest/RenderScaleAutoTest.java b/src/main/java/dev/zelo/renderscale/gametest/RenderScaleAutoTest.java index 228cf2e..97f3af6 100644 --- a/src/main/java/dev/zelo/renderscale/gametest/RenderScaleAutoTest.java +++ b/src/main/java/dev/zelo/renderscale/gametest/RenderScaleAutoTest.java @@ -223,7 +223,7 @@ private void setPhase(Phase next) { phaseFrames = 0; } - private void createWorld(Minecraft client) { + void createWorld(Minecraft client) { LevelSettings levelSettings = new LevelSettings( "RenderScale Autotest", GameType.CREATIVE, diff --git a/src/main/java/dev/zelo/renderscale/gametest/RenderScaleClientGameTest.java b/src/main/java/dev/zelo/renderscale/gametest/RenderScaleClientGameTest.java index 5d6fbde..d45d1ad 100644 --- a/src/main/java/dev/zelo/renderscale/gametest/RenderScaleClientGameTest.java +++ b/src/main/java/dev/zelo/renderscale/gametest/RenderScaleClientGameTest.java @@ -49,7 +49,7 @@ public void runTest(ClientGameTestContext context) { Path scaledShot = context.takeScreenshot("renderscale_scaled"); ScreenshotVerifier.verifyScaling(nativeShot, scaledShot); - //? >=26.3 + //? >=26.2 verifyFilters(context, nativeShot); verifyDynamicScale(context); //? >=26.3 @@ -85,7 +85,7 @@ private static void verifyWindowResize(ClientGameTestContext context) { } //?} - //? >=26.3 { + //? >=26.2 { private static void verifyFilters(ClientGameTestContext context, Path nativeShot) { try { setRenderScale(context, TEST_SCALE); @@ -98,14 +98,31 @@ private static void verifyFilters(ClientGameTestContext context, Path nativeShot context.runOnClient(client -> { // Fail the test directly if ShaderC cannot compile either FSR pass. + //? >=26.3 { com.mojang.blaze3d.systems.RenderSystem.getCompiledPipeline(RenderScale.FSR_EASU_PIPELINE); com.mojang.blaze3d.systems.RenderSystem.getCompiledPipeline(RenderScale.FSR_RCAS_PIPELINE); + //?} else { + /*var device = com.mojang.blaze3d.systems.RenderSystem.getDevice(); + if (!device.precompilePipeline(RenderScale.FSR_EASU_PIPELINE).isValid() + || !device.precompilePipeline(RenderScale.FSR_RCAS_PIPELINE).isValid()) { + throw new AssertionError("FSR pipelines must compile"); + } + *///?} RenderScale.getConfig().forceLinear = false; RenderScale.getConfig().fsr = true; RenderScale.CONFIG.save(); }); context.waitTicks(5); ScreenshotVerifier.verifyFilteredScaling(nativeShot, context.takeScreenshot("renderscale_fsr")); + + // Don't need to test this... +// setRenderScale(context, 1.0f); +// context.waitTicks(5); +// ScreenshotVerifier.verifyFilteredScaling(nativeShot, context.takeScreenshot("renderscale_fsr_native")); +// +// setRenderScale(context, 2.0f); +// context.waitTicks(5); +// ScreenshotVerifier.verifyFilteredScaling(nativeShot, context.takeScreenshot("renderscale_fsr_downsampled")); } finally { context.runOnClient(client -> { RenderScale.getConfig().fsr = false; diff --git a/src/main/java/dev/zelo/renderscale/mixin/MixinDebugScreenOverlay.java b/src/main/java/dev/zelo/renderscale/mixin/MixinDebugScreenOverlay.java index 81d53e6..318ddfa 100644 --- a/src/main/java/dev/zelo/renderscale/mixin/MixinDebugScreenOverlay.java +++ b/src/main/java/dev/zelo/renderscale/mixin/MixinDebugScreenOverlay.java @@ -67,8 +67,8 @@ public void display(DebugScreenDisplayer displayer, Level level, LevelChunk clientChunk, LevelChunk serverChunk) { RenderScale renderScale = RenderScale.getInstance(); if (renderScale != null) { - displayer.addPriorityLine(String.format(Locale.ROOT, "Render scale: %.1f%%", - renderScale.getRenderScaleFactor() * 100.0)); + displayer.addPriorityLine(String.format(Locale.ROOT, "Render scale: %.1f%% (%s)", + renderScale.getRenderScaleFactor() * 100.0, renderScale.getScalingMode())); } } }; @@ -101,8 +101,8 @@ public void display(DebugScreenDisplayer displayer, Level level, RenderScale renderScale = RenderScale.getInstance(); if (renderScale != null) { List lines = new ArrayList<>(callbackInfo.getReturnValue()); - lines.add(Math.min(2, lines.size()), String.format(Locale.ROOT, "Render scale: %.1f%%", - renderScale.getRenderScaleFactor() * 100.0)); + lines.add(Math.min(2, lines.size()), String.format(Locale.ROOT, "Render scale: %.1f%% (%s)", + renderScale.getRenderScaleFactor() * 100.0, renderScale.getScalingMode())); callbackInfo.setReturnValue(lines); } } diff --git a/src/main/java/dev/zelo/renderscale/mixin/MixinGlPipelineRecompiler.java b/src/main/java/dev/zelo/renderscale/mixin/MixinGlPipelineRecompiler.java new file mode 100644 index 0000000..27089ca --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/mixin/MixinGlPipelineRecompiler.java @@ -0,0 +1,26 @@ +//? >=26.3 { +package dev.zelo.renderscale.mixin; + +import com.mojang.renderpearl.backend.api.BackendRenderPipeline; +import com.mojang.renderpearl.backend.opengl.GlPipelineRecompiler; +import dev.kikugie.fletching_table.annotation.MixinEnvironment; +import org.spongepowered.asm.mixin.Mixin; +import org.spongepowered.asm.mixin.injection.Constant; +import org.spongepowered.asm.mixin.injection.ModifyConstant; + +@Mixin(GlPipelineRecompiler.class) +@MixinEnvironment(type = MixinEnvironment.Env.CLIENT) +public abstract class MixinGlPipelineRecompiler { + @ModifyConstant(method = "decompileShader", constant = @Constant(intValue = 330)) + private int renderScale$fp16ShaderVersion(int version, BackendRenderPipeline.CreateInfo.Shader shader) { + // RenderPearl otherwise lowers even GLSL 450 input to 330, where the + // AMD half-float built-in overloads are unavailable. These two shaders + // are only selected after checking FP16 extensions; pipeline compilation + // verifies that the driver accepts this GLSL version before using FP16. + return switch (shader.name()) { + case "renderscale:core/easu_fp16", "renderscale:core/rcas_fp16" -> 450; + default -> version; + }; + } +} +//?} diff --git a/src/main/java/dev/zelo/renderscale/mixin/MixinVulkanBackend.java b/src/main/java/dev/zelo/renderscale/mixin/MixinVulkanBackend.java new file mode 100644 index 0000000..e953cbf --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/mixin/MixinVulkanBackend.java @@ -0,0 +1,68 @@ +//? >=26.2 { +package dev.zelo.renderscale.mixin; + +//? >=26.3 { +import com.llamalad7.mixinextras.injector.ModifyExpressionValue; +import com.mojang.renderpearl.backend.vulkan.VulkanBackend; +import com.mojang.renderpearl.backend.vulkan.VulkanPhysicalDevice; +import com.mojang.renderpearl.backend.vulkan.init.FeatureSet; +//?} else { +/*import com.mojang.blaze3d.vulkan.VulkanBackend; +import com.mojang.blaze3d.vulkan.VulkanPhysicalDevice; +import com.mojang.blaze3d.vulkan.init.VulkanFeature; +import org.lwjgl.system.MemoryStack; +import org.lwjgl.vulkan.VK11; +import org.lwjgl.vulkan.VkPhysicalDeviceFeatures2; +import org.lwjgl.vulkan.VkPhysicalDeviceVulkan12Features; +import java.util.Collection; +*///?} +import dev.kikugie.fletching_table.annotation.MixinEnvironment; +import dev.zelo.renderscale.compat.vulkan.VulkanFsrSupport; +import org.lwjgl.vulkan.VkDevice; +import org.spongepowered.asm.mixin.Mixin; +import org.spongepowered.asm.mixin.injection.At; +import org.spongepowered.asm.mixin.injection.Inject; +import org.spongepowered.asm.mixin.injection.callback.CallbackInfoReturnable; + +import java.util.HashSet; +import java.util.Set; + +@Mixin(VulkanBackend.class) +@MixinEnvironment(type = MixinEnvironment.Env.CLIENT) +public abstract class MixinVulkanBackend { + //? >=26.3 { + @ModifyExpressionValue(method = "createDevice", at = @At(value = "INVOKE", + target = "Lcom/mojang/renderpearl/backend/vulkan/VulkanFeatureSets;optionalFeatureSets()Ljava/util/Set;")) + private Set renderScale$optionalFp16(Set original) { + Set features = new HashSet<>(original); + features.add(VulkanFsrSupport.FP16); + return features; + } + + @Inject(method = "createDevice(Lcom/mojang/renderpearl/backend/vulkan/init/FeatureSet;Lcom/mojang/renderpearl/backend/vulkan/VulkanPhysicalDevice;)Lorg/lwjgl/vulkan/VkDevice;", at = @At("RETURN")) + private static void renderScale$recordFp16(FeatureSet features, VulkanPhysicalDevice physicalDevice, + CallbackInfoReturnable cir) { + VulkanFsrSupport.recordEnabled(cir.getReturnValue(), features.contains(VulkanFsrSupport.FP16)); + } + //?} else { + /*@Inject(method = "createDevice(Ljava/util/Collection;Lcom/mojang/blaze3d/vulkan/VulkanPhysicalDevice;Ljava/util/Set;)Lorg/lwjgl/vulkan/VkDevice;", at = @At("HEAD")) + private static void renderScale$optionalFp16(Collection extensions, VulkanPhysicalDevice physicalDevice, + Set features, CallbackInfoReturnable cir) { + // 26.2 predates optional FeatureSets. Query the selected GPU and add only + // supported FP16 arithmetic to its per-device feature set. + try (MemoryStack stack = MemoryStack.stackPush()) { + var fp16 = VkPhysicalDeviceVulkan12Features.calloc(stack).sType$Default(); + var supported = VkPhysicalDeviceFeatures2.calloc(stack).sType$Default().pNext(fp16); + VK11.vkGetPhysicalDeviceFeatures2(physicalDevice.vkPhysicalDevice(), supported); + if (fp16.shaderFloat16()) features.add(VulkanFsrSupport.SHADER_FLOAT16); + } + } + + @Inject(method = "createDevice(Ljava/util/Collection;Lcom/mojang/blaze3d/vulkan/VulkanPhysicalDevice;Ljava/util/Set;)Lorg/lwjgl/vulkan/VkDevice;", at = @At("RETURN")) + private static void renderScale$recordFp16(Collection extensions, VulkanPhysicalDevice physicalDevice, + Set features, CallbackInfoReturnable cir) { + VulkanFsrSupport.recordEnabled(cir.getReturnValue(), features.contains(VulkanFsrSupport.SHADER_FLOAT16)); + } + *///?} +} +//?} diff --git a/src/main/java/dev/zelo/renderscale/mixin/accessors/MixinGpuDeviceAccessor.java b/src/main/java/dev/zelo/renderscale/mixin/accessors/MixinGpuDeviceAccessor.java new file mode 100644 index 0000000..aed3441 --- /dev/null +++ b/src/main/java/dev/zelo/renderscale/mixin/accessors/MixinGpuDeviceAccessor.java @@ -0,0 +1,24 @@ +//? >=26.2 { +package dev.zelo.renderscale.mixin.accessors; + +//? >=26.3 { +import com.mojang.renderpearl.backend.api.GpuDeviceBackend; +import com.mojang.renderpearl.frontend.FrontendGpuDevice; +//?} else { +/*import com.mojang.blaze3d.systems.GpuDevice; +import com.mojang.blaze3d.systems.GpuDeviceBackend; +*///?} +import dev.kikugie.fletching_table.annotation.MixinEnvironment; +import org.spongepowered.asm.mixin.Mixin; +import org.spongepowered.asm.mixin.gen.Accessor; + +//? >=26.3 { +@Mixin(FrontendGpuDevice.class) +//?} else +//@Mixin(GpuDevice.class) +@MixinEnvironment(type = MixinEnvironment.Env.CLIENT) +public interface MixinGpuDeviceAccessor { + @Accessor("backend") + GpuDeviceBackend renderScale$getBackend(); +} +//?} diff --git a/src/main/resources/assets/renderscale/lang/en_us.json b/src/main/resources/assets/renderscale/lang/en_us.json index 28b095c..3dba030 100644 --- a/src/main/resources/assets/renderscale/lang/en_us.json +++ b/src/main/resources/assets/renderscale/lang/en_us.json @@ -10,9 +10,9 @@ "text.autoconfig.renderscale.option.aggression.@Tooltip": "How quickly resolution changes (Note: Resolution only changes when the camera moves!).", "text.autoconfig.renderscale.option.minimumScale": "Minimum scale", "text.autoconfig.renderscale.option.minimumScale.@Tooltip": "The lowest scale that will be set.", - "text.autoconfig.renderscale.option.fsr": "Use FSR 1 (Experimental)", - "text.autoconfig.renderscale.option.fsr.@Tooltip": "Results may vary depending on your hardware. Compare your FPS!", - "text.autoconfig.renderscale.option.fsr.@Tooltip.sodium": "Results may vary depending on your hardware, compare your FPS!", + "text.autoconfig.renderscale.option.fsr": "Use FSR 1", + "text.autoconfig.renderscale.option.fsr.@Tooltip": "Results vary depending on your hardware. Compare your FPS!", + "text.autoconfig.renderscale.option.fsr.@Tooltip.sodium": "Results vary depending on your hardware, compare your FPS!", "text.autoconfig.renderscale.option.downscaleFilter": "Supersampling filter", "text.autoconfig.renderscale.option.downscaleFilter.@Tooltip": "How the higher-resolution 3D view is downscaled to the window. Rotated Grid is best around 2x scale. Sparse Grid is a wider filter used at 300% scale and above; at lower scales it falls back to Rotated Grid. Applies when the scale is above 100%.", "text.autoconfig.renderscale.option.downscaleFilter.BILINEAR": "Bilinear (default)", diff --git a/src/main/resources/assets/renderscale/shaders/core/easu.fsh b/src/main/resources/assets/renderscale/shaders/core/easu.fsh index f19e1b2..31495c5 100644 --- a/src/main/resources/assets/renderscale/shaders/core/easu.fsh +++ b/src/main/resources/assets/renderscale/shaders/core/easu.fsh @@ -1,93 +1,21 @@ #version 420 -//#version 150 #define A_GPU 1 #define A_GLSL 1 -// Use the portable 32-bit implementation initially. #define FSR_EASU_F 1 +#define FSR_COLOUR AF3 +#define FSR_GATHER AF4 +#define FSR_EASU_R FsrEasuRF +#define FSR_EASU_G FsrEasuGF +#define FSR_EASU_B FsrEasuBF +#define FSR_EASU FsrEasuF -#moj_import #moj_import +#moj_import +#moj_import uniform sampler2D InSampler; - out vec4 fragColor; -#ifdef RENDERSCALE_EXPLICIT_GATHER -// RenderPearl translates OpenGL shaders to GLSL 330, where SPIRV-Cross cannot -// emit textureGather with a nonzero component. Match its four texels explicitly. -vec4 renderScaleGather(vec2 p, int component) -{ - ivec2 size = textureSize(InSampler, 0); - ivec2 base = ivec2(floor(p * vec2(size) - 0.5)); - ivec2 maximum = size - ivec2(1); - return vec4( - texelFetch(InSampler, clamp(base + ivec2(0, 1), ivec2(0), maximum), 0)[component], - texelFetch(InSampler, clamp(base + ivec2(1, 1), ivec2(0), maximum), 0)[component], - texelFetch(InSampler, clamp(base + ivec2(1, 0), ivec2(0), maximum), 0)[component], - texelFetch(InSampler, clamp(base, ivec2(0), maximum), 0)[component] - ); -} -#define RENDERSCALE_GATHER(p, component) renderScaleGather(p, component) -#else -#define RENDERSCALE_GATHER(p, component) textureGather(InSampler, p, component) -#endif - -/* - * These callbacks must exist before importing ffx_fsr1.glsl. - * - * Minecraft samplers are normally combined sampler2D objects, unlike - * AMD's Vulkan example, which declares separate texture and sampler objects. - */ -AF4 FsrEasuRF(AF2 p) -{ -return RENDERSCALE_GATHER(p, 0); -} - -AF4 FsrEasuGF(AF2 p) -{ -return RENDERSCALE_GATHER(p, 1); -} - -AF4 FsrEasuBF(AF2 p) -{ -return RENDERSCALE_GATHER(p, 2); -} - -#moj_import - -void main() -{ -AU2 outputPixel = AU2(gl_FragCoord.xy); -vec2 inputSize = vec2(textureSize(InSampler, 0)); -AU4 fsrConst0; -AU4 fsrConst1; -AU4 fsrConst2; -AU4 fsrConst3; - -FsrEasuCon( -fsrConst0, -fsrConst1, -fsrConst2, -fsrConst3, -inputSize.x, -inputSize.y, -inputSize.x, -inputSize.y, -ScreenSize.x, -ScreenSize.y -); - -AF3 colour; -FsrEasuF( -colour, -outputPixel, -fsrConst0, -fsrConst1, -fsrConst2, -fsrConst3 -); - -fragColor = vec4(colour, 1.0); -} +#moj_import diff --git a/src/main/resources/assets/renderscale/shaders/core/easu_fp16.fsh b/src/main/resources/assets/renderscale/shaders/core/easu_fp16.fsh new file mode 100644 index 0000000..efd3d4e --- /dev/null +++ b/src/main/resources/assets/renderscale/shaders/core/easu_fp16.fsh @@ -0,0 +1,23 @@ +#version 450 + +#define A_GPU 1 +#define A_GLSL 1 + +#moj_import + +#define FSR_EASU_H 1 +#define FSR_COLOUR AH3 +#define FSR_GATHER AH4 +#define FSR_EASU_R FsrEasuRH +#define FSR_EASU_G FsrEasuGH +#define FSR_EASU_B FsrEasuBH +#define FSR_EASU FsrEasuH + +#moj_import +#moj_import +#moj_import + +uniform sampler2D InSampler; +out vec4 fragColor; + +#moj_import diff --git a/src/main/resources/assets/renderscale/shaders/core/rcas.fsh b/src/main/resources/assets/renderscale/shaders/core/rcas.fsh index 4fe5720..193bb5e 100644 --- a/src/main/resources/assets/renderscale/shaders/core/rcas.fsh +++ b/src/main/resources/assets/renderscale/shaders/core/rcas.fsh @@ -1,69 +1,21 @@ #version 420 -//#version 150 #define A_GPU 1 #define A_GLSL 1 + #define FSR_RCAS_F 1 +#define FSR_COLOUR AF3 +#define FSR_SCALAR AF1 +#define FSR_LOAD AF4 +#define FSR_POSITION ASU2 +#define FSR_RCAS_LOAD FsrRcasLoadF +#define FSR_RCAS_INPUT FsrRcasInputF +#define FSR_RCAS FsrRcasF #moj_import +#moj_import uniform sampler2D InSampler; - out vec4 fragColor; -AF4 FsrRcasLoadF(ASU2 p) -{ -ivec2 size = textureSize(InSampler, 0); - -// RCAS samples one pixel outside the current location at image edges. -ivec2 position = clamp(ivec2(p), ivec2(0), size - ivec2(1)); - -return texelFetch(InSampler, position, 0); -} - -void FsrRcasInputF(inout AF1 r, inout AF1 g, inout AF1 b) -{ -// The required input is normalized, display-encoded (sRGB-style -// perceptual) RGB, which is what vanilla is expected to render into the -// main target. This comment is not verification of every attachment -// format or shader-pack colour path: do not add a colour conversion here -// without checking the actual format in game. -} - -#moj_import - -// The fixed sharpness tuning is AMD's shipped FSR1 default (stop parameter -// 0.2, i.e. gain 2^-0.2); the FSR pipeline relies on that default. -#ifndef RENDERSCALE_RCAS_GAIN -#define RENDERSCALE_RCAS_GAIN exp2(-0.2) -#endif - -void main() -{ -AU2 outputPixel = AU2(gl_FragCoord.xy); -// FsrRcasF only uses the 32-bit sharpness in con.x. Avoid generating the unused -// half-float constants, which require packing functions absent from GLSL 330. -AU4 fsrConst0 = AU4(floatBitsToUint(RENDERSCALE_RCAS_GAIN), 0u, 0u, 0u); - -AF3 colour; -FsrRcasF( -colour.r, -colour.g, -colour.b, -outputPixel, -fsrConst0 -); - -// Supersampling pipelines define RENDERSCALE_SHARPNESS (0-1) and blend -// towards the unsharpened centre pixel so the strength control is a -// consistent mix; the 0% setting is skipped CPU-side, since RCAS's fast -// normalization is not an exact passthrough. The Noise-protected variant -// additionally sets FSR_RCAS_DENOISE. The FSR upscale pipeline defines -// neither, so its output path is unchanged. -#ifdef RENDERSCALE_SHARPNESS -AF3 centre = texelFetch(InSampler, ivec2(outputPixel), 0).rgb; -fragColor = vec4(mix(centre, colour, RENDERSCALE_SHARPNESS), 1.0); -#else -fragColor = vec4(colour, 1.0); -#endif -} +#moj_import diff --git a/src/main/resources/assets/renderscale/shaders/core/rcas_fp16.fsh b/src/main/resources/assets/renderscale/shaders/core/rcas_fp16.fsh new file mode 100644 index 0000000..d428a18 --- /dev/null +++ b/src/main/resources/assets/renderscale/shaders/core/rcas_fp16.fsh @@ -0,0 +1,23 @@ +#version 450 + +#define A_GPU 1 +#define A_GLSL 1 + +#moj_import + +#define FSR_RCAS_H 1 +#define FSR_COLOUR AH3 +#define FSR_SCALAR AH1 +#define FSR_LOAD AH4 +#define FSR_POSITION ASW2 +#define FSR_RCAS_LOAD FsrRcasLoadH +#define FSR_RCAS_INPUT FsrRcasInputH +#define FSR_RCAS FsrRcasH + +#moj_import +#moj_import + +uniform sampler2D InSampler; +out vec4 fragColor; + +#moj_import diff --git a/src/main/resources/assets/renderscale/shaders/include/fsr_easu.glsl b/src/main/resources/assets/renderscale/shaders/include/fsr_easu.glsl new file mode 100644 index 0000000..95238a8 --- /dev/null +++ b/src/main/resources/assets/renderscale/shaders/include/fsr_easu.glsl @@ -0,0 +1,75 @@ +#ifdef RENDERSCALE_EXPLICIT_GATHER +// RenderPearl translates OpenGL shaders to GLSL 330, where SPIRV-Cross cannot +// emit textureGather with a nonzero component. Match its four texels explicitly. +vec4 renderScaleGather(vec2 p, int component) +{ + ivec2 size = textureSize(InSampler, 0); + ivec2 base = ivec2(floor(p * vec2(size) - 0.5)); + ivec2 maximum = size - ivec2(1); + return vec4( + texelFetch(InSampler, clamp(base + ivec2(0, 1), ivec2(0), maximum), 0)[component], + texelFetch(InSampler, clamp(base + ivec2(1, 1), ivec2(0), maximum), 0)[component], + texelFetch(InSampler, clamp(base + ivec2(1, 0), ivec2(0), maximum), 0)[component], + texelFetch(InSampler, clamp(base, ivec2(0), maximum), 0)[component] + ); +} +#define RENDERSCALE_GATHER(p, component) renderScaleGather(p, component) +#else +#define RENDERSCALE_GATHER(p, component) textureGather(InSampler, p, component) +#endif + +/* + * These callbacks implement the prototypes declared by ffx_fsr1.glsl. + * + * Minecraft samplers are normally combined sampler2D objects, unlike + * AMD's Vulkan example, which declares separate texture and sampler objects. + */ +FSR_GATHER FSR_EASU_R(AF2 p) +{ +return FSR_GATHER(RENDERSCALE_GATHER(p, 0)); +} + +FSR_GATHER FSR_EASU_G(AF2 p) +{ +return FSR_GATHER(RENDERSCALE_GATHER(p, 1)); +} + +FSR_GATHER FSR_EASU_B(AF2 p) +{ +return FSR_GATHER(RENDERSCALE_GATHER(p, 2)); +} + +void main() +{ +AU2 outputPixel = AU2(gl_FragCoord.xy); +vec2 inputSize = vec2(textureSize(InSampler, 0)); +AU4 fsrConst0; +AU4 fsrConst1; +AU4 fsrConst2; +AU4 fsrConst3; + +FsrEasuCon( +fsrConst0, +fsrConst1, +fsrConst2, +fsrConst3, +inputSize.x, +inputSize.y, +inputSize.x, +inputSize.y, +ScreenSize.x, +ScreenSize.y +); + +FSR_COLOUR colour; +FSR_EASU( +colour, +outputPixel, +fsrConst0, +fsrConst1, +fsrConst2, +fsrConst3 +); + +fragColor = vec4(colour, 1.0); +} diff --git a/src/main/resources/assets/renderscale/shaders/include/fsr_precision.glsl b/src/main/resources/assets/renderscale/shaders/include/fsr_precision.glsl new file mode 100644 index 0000000..b745630 --- /dev/null +++ b/src/main/resources/assets/renderscale/shaders/include/fsr_precision.glsl @@ -0,0 +1,55 @@ +// Set by the renderer only after checking the active device's capabilities. +#ifdef RENDERSCALE_FP16 +#if RENDERSCALE_FP16 == 1 +#extension GL_EXT_shader_explicit_arithmetic_types_float16 : require +#elif RENDERSCALE_FP16 == 2 +#extension GL_AMD_gpu_shader_half_float : require +#elif RENDERSCALE_FP16 == 3 +#extension GL_NV_gpu_shader5 : require +#endif + +#define A_HALF 1 +#define A_SKIP_EXT 1 + +// FSR needs FP16 arithmetic, but not 16-bit storage or integer arithmetic. +// Keep pixel coordinates and bit manipulation at 32 bits, including on GPUs +// that expose half floats without int16. Packing preserves the half bit pattern. +#define uint16_t uint +#define u16vec2 uvec2 +#define u16vec3 uvec3 +#define u16vec4 uvec4 +#define int16_t int +#define i16vec2 ivec2 +#define i16vec3 ivec3 +#define i16vec4 ivec4 + +#define packUint2x16(value) (((value).x & 0xffffu) | (((value).y & 0xffffu) << 16u)) +#define unpackUint2x16(value) (uvec2(value, (value) >> 16u) & uvec2(0xffffu)) + +uint renderScaleHalfBits(float16_t value) { + return packFloat2x16(f16vec2(value, float16_t(0.0))) & 0xffffu; +} +float16_t renderScaleBitsHalf(uint value) { + return unpackFloat2x16(value & 0xffffu).x; +} +uvec2 renderScaleHalfBits(f16vec2 value) { + return uvec2(renderScaleHalfBits(value.x), renderScaleHalfBits(value.y)); +} +f16vec2 renderScaleBitsHalf(uvec2 value) { + return f16vec2(renderScaleBitsHalf(value.x), renderScaleBitsHalf(value.y)); +} +uvec3 renderScaleHalfBits(f16vec3 value) { + return uvec3(renderScaleHalfBits(value.x), renderScaleHalfBits(value.y), renderScaleHalfBits(value.z)); +} +f16vec3 renderScaleBitsHalf(uvec3 value) { + return f16vec3(renderScaleBitsHalf(value.x), renderScaleBitsHalf(value.y), renderScaleBitsHalf(value.z)); +} +uvec4 renderScaleHalfBits(f16vec4 value) { + return uvec4(renderScaleHalfBits(value.x), renderScaleHalfBits(value.y), renderScaleHalfBits(value.z), renderScaleHalfBits(value.w)); +} +f16vec4 renderScaleBitsHalf(uvec4 value) { + return f16vec4(renderScaleBitsHalf(value.x), renderScaleBitsHalf(value.y), renderScaleBitsHalf(value.z), renderScaleBitsHalf(value.w)); +} +#define halfBitsToUint16 renderScaleHalfBits +#define uint16BitsToHalf renderScaleBitsHalf +#endif diff --git a/src/main/resources/assets/renderscale/shaders/include/fsr_rcas.glsl b/src/main/resources/assets/renderscale/shaders/include/fsr_rcas.glsl new file mode 100644 index 0000000..276d4d1 --- /dev/null +++ b/src/main/resources/assets/renderscale/shaders/include/fsr_rcas.glsl @@ -0,0 +1,60 @@ +FSR_LOAD FSR_RCAS_LOAD(FSR_POSITION p) +{ +ivec2 size = textureSize(InSampler, 0); + +// RCAS samples one pixel outside the current location at image edges. +ivec2 position = clamp(ivec2(p), ivec2(0), size - ivec2(1)); + +return FSR_LOAD(texelFetch(InSampler, position, 0)); +} + +void FSR_RCAS_INPUT(inout FSR_SCALAR r, inout FSR_SCALAR g, inout FSR_SCALAR b) +{ +// The required input is normalized, display-encoded (sRGB-style +// perceptual) RGB, which is what vanilla is expected to render into the +// main target. This comment is not verification of every attachment +// format or shader-pack colour path: do not add a colour conversion here +// without checking the actual format in game. +} + +// The fixed sharpness tuning is AMD's shipped FSR1 default (stop parameter +// 0.2, i.e. gain 2^-0.2); the FSR pipeline relies on that default. +#ifndef RENDERSCALE_RCAS_GAIN +#define RENDERSCALE_RCAS_GAIN exp2(-0.2) +#endif + +void main() +{ +AU2 outputPixel = AU2(gl_FragCoord.xy); +#ifdef A_HALF +// RCAS H reads the packed half sharpness from con.y, unlike RCAS F's con.x. +AF1 sharpness = RENDERSCALE_RCAS_GAIN; +AU4 fsrConst0 = AU4(floatBitsToUint(sharpness), packFloat2x16(AH2(sharpness)), 0u, 0u); +#else +// FsrRcasF only uses the 32-bit sharpness in con.x. Avoid generating the unused +// half-float constants, which require packing functions absent from GLSL 330. +AU4 fsrConst0 = AU4(floatBitsToUint(RENDERSCALE_RCAS_GAIN), 0u, 0u, 0u); +#endif + +FSR_COLOUR colour; +FSR_RCAS( +colour.r, +colour.g, +colour.b, +outputPixel, +fsrConst0 +); + +// Supersampling pipelines define RENDERSCALE_SHARPNESS (0-1) and blend +// towards the unsharpened centre pixel so the strength control is a +// consistent mix; the 0% setting is skipped CPU-side, since RCAS's fast +// normalization is not an exact passthrough. The Noise-protected variant +// additionally sets FSR_RCAS_DENOISE. The FSR upscale pipeline defines +// neither, so its output path is unchanged. +#ifdef RENDERSCALE_SHARPNESS +AF3 centre = texelFetch(InSampler, ivec2(outputPixel), 0).rgb; +fragColor = vec4(mix(centre, AF3(colour), RENDERSCALE_SHARPNESS), 1.0); +#else +fragColor = vec4(colour, 1.0); +#endif +} diff --git a/src/test/python/test_fsr_shaders.py b/src/test/python/test_fsr_shaders.py new file mode 100644 index 0000000..887c993 --- /dev/null +++ b/src/test/python/test_fsr_shaders.py @@ -0,0 +1,86 @@ +#!/usr/bin/env python3 +"""Compile FSR and supersampling RCAS variants; requires glslangValidator, spirv-dis and spirv-val. + +NV_gpu_shader5 must be tested on an NVIDIA driver: glslang does not support it. +The EXT and AMD variants exercise the same half-float compatibility helpers. +""" +from pathlib import Path +import re +import shutil +import subprocess +import tempfile + +ROOT = Path(__file__).resolve().parents[3] / "src/main/resources/assets/renderscale/shaders" + + +def expand_imports(source): + def include(match): + if match[1] == "minecraft:globals.glsl": + return "layout(std140, binding=0) uniform Globals { vec2 ScreenSize; };" + included = (ROOT / "include" / match[1].split(":")[1]).read_text() + # RenderPearl rewrites moj_import and shader interfaces only in the root source. + assert "#moj_import" not in included, match[1] + assert "uniform sampler" not in included and "out vec4 fragColor" not in included, match[1] + return included + + return re.sub(r"#moj_import <([^>]+)>", include, source) + + +def main(): + for tool in ("glslangValidator", "spirv-dis", "spirv-val"): + if not shutil.which(tool): + raise SystemExit(f"Missing required tool: {tool}") + variants = [("easu", None, False), ("rcas", None, False)] + variants += [("rcas", strength, denoise) + for strength in (0.0, 0.5, 1.0) for denoise in (False, True)] + with tempfile.TemporaryDirectory(prefix="renderscale-fsr-") as directory: + for shader, strength, denoise in variants: + for precision, extension in (("fp32", 0), ("fp16-ext", 1), ("fp16-amd", 2)): + for explicit_gather in (False, True): + label = f"{shader}-{precision}-{'explicit' if explicit_gather else 'native'}" + if strength is not None: + label += f"-sharpen-{strength}" + ("-denoise" if denoise else "") + suffix = "_fp16" if extension else "" + source = expand_imports((ROOT / "core" / f"{shader}{suffix}.fsh").read_text()) + defines = f"\n#define RENDERSCALE_FP16 {extension}" if extension else "" + if explicit_gather: + defines += "\n#define RENDERSCALE_EXPLICIT_GATHER" + if strength is not None: + defines += f"\n#define RENDERSCALE_SHARPNESS {strength}" + if denoise: + defines += "\n#define FSR_RCAS_DENOISE" + version, body = source.split("\n", 1) + source = version + defines + "\n" + body + path = Path(directory) / f"{label}.frag" + binary = path.with_suffix(".spv") + path.write_text(source) + for target in ("opengl", "vulkan"): + if target == "vulkan" and extension == 2: + continue # Vulkan uses the EXT arithmetic shader variant. + target_label = f"{target}-{label}" + flags = ["-G"] if target == "opengl" else ["-V", "--target-env", "vulkan1.2"] + subprocess.run([ + "glslangValidator", *flags, "--auto-map-bindings", "--auto-map-locations", + "-o", str(binary), str(path), + ], check=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) + validation = [] if target == "opengl" else ["--target-env", "vulkan1.2"] + subprocess.run(["spirv-val", *validation, str(binary)], check=True, + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) + assembly = subprocess.check_output(["spirv-dis", str(binary)], text=True) + # Require real half arithmetic with no int16 or storage features; + # fallback must not require the Vulkan shaderFloat16 feature. + assert ("OpCapability Float16" in assembly) == bool(extension), target_label + assert ("OpTypeFloat 16" in assembly) == bool(extension), target_label + assert "OpCapability Int16" not in assembly, target_label + assert "OpCapability Storage" not in assembly, target_label + if shader == "rcas": + # The shared include must retain supersampling's strength blend. + assert (" FMix " in assembly) == (strength is not None), target_label + print(f"PASS {target_label}") + + +if __name__ == "__main__": + try: + main() + except subprocess.CalledProcessError as error: + raise SystemExit(error.stdout or str(error)) diff --git a/stonecutter.gradle.kts b/stonecutter.gradle.kts index f967618..663d064 100644 --- a/stonecutter.gradle.kts +++ b/stonecutter.gradle.kts @@ -28,6 +28,7 @@ stonecutter parameters { constants["sodium"] = current.parsed >= "1.21.11" || current.version == "1.21.1" // Versions affected by stuttering with Dynamic Scaling + Iris Shaders - constants["stutter"] = current.version == "1.20.1" || current.version == "1.21.1" + // TODO: Looks like all versions have this issue :( + constants["stutter"] = true//current.version == "1.20.1" || current.version == "1.21.1" }