aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorhachem <im@hachem.wtf>2026-08-22 21:40:20 +0200
committerhachem <im@hachem.wtf>2026-08-22 21:40:20 +0200
commit43cee69e40be3cb8b0341c3fc94171fe01712b8c (patch)
treeb5a34b0aa9226c421c617da675275acab2919558
parentb7d792c553bf64ad39acff56039a638947f7165b (diff)
[feat]: mip sampling
-rw-r--r--assets/shaders/Geodesic.slang87
-rw-r--r--src/platform/vulkan/vulkan_renderer.cpp52
2 files changed, 112 insertions, 27 deletions
diff --git a/assets/shaders/Geodesic.slang b/assets/shaders/Geodesic.slang
index a0d030e..9120ac0 100644
--- a/assets/shaders/Geodesic.slang
+++ b/assets/shaders/Geodesic.slang
@@ -83,11 +83,6 @@ struct Hit
float hitRadius;
};
-float3 SampleHDRI(float3 direction)
-{
- return u_HDRIEnvironment.Sample(direction).rgb;
-}
-
float hash(float3 p)
{
p = frac(p * float3(0.1031, 0.1030, 0.0973));
@@ -307,6 +302,13 @@ float CalculateAdaptiveStepSize(Ray ray, float baseStepSize)
// region near the photon sphere is resolved with tiny ones. This keeps the
// integration accurate near the hole regardless of how far the camera is.
float step = 0.02 * max(ray.r - R_PHOTON, 0.0);
+
+ // Slow down when near the disk plane (within its radial extent) so the thin
+ // slab is never stepped over -- otherwise grazing rays leak through it.
+ float rc = length(float2(ray.x, ray.z));
+ if (rc < disk.disk_r2 * 3.0 && abs(ray.y) < disk.thickness * 8.0)
+ step = min(step, disk.thickness);
+
return clamp(step, MIN_STEP_SIZE, MAX_STEP_SIZE);
}
@@ -315,11 +317,12 @@ float3 ACESFilm(float3 x)
return clamp((x * (2.51 * x + 0.03)) / (x * (2.43 * x + 0.59) + 0.14), 0.0, 1.0);
}
-[shader("fragment")]
-float4 fragmentMain(VSOutput input) : SV_Target
+// Trace one primary ray for the given image UV and return its linear,
+// pre-tone-map radiance. Called once per sub-sample by fragmentMain.
+float3 TracePixel(float2 texCoord)
{
- float u = (2.0 * input.texCoord.x - 1.0) * cam.aspect * cam.tanHalfFov;
- float v = (1.0 - 2.0 * input.texCoord.y) * cam.tanHalfFov;
+ float u = (2.0 * texCoord.x - 1.0) * cam.aspect * cam.tanHalfFov;
+ float v = (1.0 - 2.0 * texCoord.y) * cam.tanHalfFov;
float3 dir = normalize(u * cam.camRight - v * cam.camUp + cam.camForward);
Ray ray = InitRay(cam.camPos, dir);
@@ -350,20 +353,26 @@ float4 fragmentMain(VSOutput input) : SV_Target
RK4Step(ray, stepSize);
float3 newPos = float3(ray.x, ray.y, ray.z);
- // Opaque thin disk: a sign change in y means the ray pierced the disk
- // plane (y = 0). The first crossing inside the annulus is a solid,
- // self-luminous surface -- it emits and blocks everything behind it, so
- // the ray stops here (near side occludes far side / background).
- if (prevPos.y * newPos.y < 0.0)
+ // Opaque disk of small half-thickness H (a slab about the midplane y=0).
+ // The ray hits when it first crosses the midplane OR enters the slab
+ // while grazing along it. Real (nonzero) thickness stops the zero-height
+ // edge-on "razor" from aliasing into a beam streaking across the frame.
{
- float t = prevPos.y / (prevPos.y - newPos.y);
- float3 cross = lerp(prevPos, newPos, t);
- float rc = length(float2(cross.x, cross.z));
- if (rc >= max(disk.disk_r1, R_ISCO) && rc <= disk.disk_r2)
+ float H = disk.thickness;
+ bool crossed = prevPos.y * newPos.y < 0.0;
+ bool inSlab = abs(newPos.y) <= H;
+ if (crossed || inSlab)
{
- diskColor = DiskEmission(cross, newPos - prevPos);
- hitDisk = true;
- break;
+ float3 hitP = crossed
+ ? lerp(prevPos, newPos, prevPos.y / (prevPos.y - newPos.y))
+ : newPos;
+ float rc = length(float2(hitP.x, hitP.z));
+ if (rc >= max(disk.disk_r1, R_ISCO) && rc <= disk.disk_r2)
+ {
+ diskColor = DiskEmission(hitP, newPos - prevPos);
+ hitDisk = true;
+ break;
+ }
}
}
@@ -374,6 +383,14 @@ float4 fragmentMain(VSOutput input) : SV_Target
if (ray.dr > 0.0 && ray.r > 50.0 * SagA_rs) break;
}
+ // Escape direction + environment mip LOD from the ray's angular divergence.
+ // Computed UNCONDITIONALLY (before the branch) so ddx/ddy are valid; strongly
+ // lensed background rays diverge fast, so they read a blurred cubemap mip and
+ // the starfield stops aliasing into a fan along the equatorial plane.
+ float3 rayDir = normalize(float3(ray.x, ray.y, ray.z) - cam.camPos);
+ float footprint = max(length(ddx(rayDir)), length(ddy(rayDir)));
+ float envLod = clamp(log2(max(footprint / 0.0015, 1.0)), 0.0, 10.0);
+
float3 shade;
if (hitDisk)
{
@@ -393,9 +410,31 @@ float4 fragmentMain(VSOutput input) : SV_Target
}
else
{
- float3 rayDir = normalize(float3(ray.x, ray.y, ray.z) - cam.camPos);
- shade = SampleHDRI(rayDir);
+ shade = u_HDRIEnvironment.SampleLevel(rayDir, envLod).rgb;
}
- return float4(ACESFilm(shade), 1.0);
+ return shade;
+}
+
+[shader("fragment")]
+float4 fragmentMain(VSOutput input) : SV_Target
+{
+ // Moving frame: one sample for responsiveness. Settled frame: rotated-grid
+ // 4x supersampling (the 4-rook pattern gives 4 distinct sub-pixel positions
+ // on BOTH axes, far better on the near-horizontal lensed edges than an
+ // ordered grid). Radiance is averaged before tone-mapping; ddx/ddy give the
+ // resolution-correct per-pixel UV footprint.
+ if (cam.moving)
+ return float4(ACESFilm(TracePixel(input.texCoord)), 1.0);
+
+ float2 dUV = float2(ddx(input.texCoord.x), ddy(input.texCoord.y));
+ float2 offs[4] = {
+ float2( 0.125, 0.375), float2( 0.375, -0.125),
+ float2(-0.125, -0.375), float2(-0.375, 0.125),
+ };
+ float3 sum = float3(0.0);
+ for (int i = 0; i < 4; ++i)
+ sum += TracePixel(input.texCoord + offs[i] * dUV);
+
+ return float4(ACESFilm(sum * 0.25), 1.0);
}
diff --git a/src/platform/vulkan/vulkan_renderer.cpp b/src/platform/vulkan/vulkan_renderer.cpp
index e19bee4..8ecf088 100644
--- a/src/platform/vulkan/vulkan_renderer.cpp
+++ b/src/platform/vulkan/vulkan_renderer.cpp
@@ -761,13 +761,17 @@ namespace Donut
const VkMemoryPropertyFlags host_vis = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
const uint32_t FACE = 1024;
const VkFormat cube_fmt = VK_FORMAT_R16G16B16A16_SFLOAT;
+ uint32_t CUBE_MIPS = 1; for (uint32_t s = FACE; s > 1; s >>= 1) ++CUBE_MIPS; // full chain (11 for 1024)
VkImageCreateInfo cci{ VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO };
cci.flags = VK_IMAGE_CREATE_CUBE_COMPATIBLE_BIT;
cci.imageType = VK_IMAGE_TYPE_2D; cci.format = cube_fmt; cci.extent = { FACE, FACE, 1 };
- cci.mipLevels = 1; cci.arrayLayers = 6; cci.samples = VK_SAMPLE_COUNT_1_BIT;
+ cci.mipLevels = CUBE_MIPS; cci.arrayLayers = 6; cci.samples = VK_SAMPLE_COUNT_1_BIT;
cci.tiling = VK_IMAGE_TILING_OPTIMAL;
- cci.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT;
+ // COLOR_ATTACHMENT renders the 6 faces (mip 0); TRANSFER_SRC/DST build the
+ // mip chain by blitting; SAMPLED for shader reads (incl. mip-LOD sampling).
+ cci.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT
+ | VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT;
VK_CHECK(vkCreateImage(device, &cci, nullptr, &cube_image));
VkMemoryRequirements creq{}; vkGetImageMemoryRequirements(device, cube_image, &creq);
VkMemoryAllocateInfo cai{ VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO };
@@ -776,10 +780,11 @@ namespace Donut
VK_CHECK(vkBindImageMemory(device, cube_image, cube_mem, 0));
VkImageViewCreateInfo cvci{ VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO };
cvci.image = cube_image; cvci.viewType = VK_IMAGE_VIEW_TYPE_CUBE; cvci.format = cube_fmt;
- cvci.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 6 };
+ cvci.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, CUBE_MIPS, 0, 6 };
VK_CHECK(vkCreateImageView(device, &cvci, nullptr, &cube_view));
VkSamplerCreateInfo csm{ VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO };
csm.magFilter = VK_FILTER_LINEAR; csm.minFilter = VK_FILTER_LINEAR;
+ csm.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; csm.minLod = 0.0f; csm.maxLod = (float)CUBE_MIPS;
csm.addressModeU = csm.addressModeV = csm.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
VK_CHECK(vkCreateSampler(device, &csm, nullptr, &cube_sampler));
@@ -984,6 +989,47 @@ namespace Donut
vkCmdDraw(cmd, 36, 1, 0, 0);
vkCmdEndRenderPass(cmd);
}
+
+ // Build the mip chain (all 6 faces) by successive linear down-blits, so
+ // divergence-based LOD sampling can read a blurred sky for lensed rays.
+ // Mip 0 (every layer) is SHADER_READ_ONLY from the render passes above.
+ {
+ VkImageMemoryBarrier src0{ VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER };
+ src0.image = cube_image; src0.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 6 };
+ src0.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; src0.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
+ src0.srcAccessMask = VK_ACCESS_SHADER_READ_BIT; src0.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
+ vkCmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &src0);
+
+ int32_t mipW = (int32_t)FACE, mipH = (int32_t)FACE;
+ for (uint32_t m = 1; m < CUBE_MIPS; ++m)
+ {
+ int32_t nW = mipW > 1 ? mipW / 2 : 1, nH = mipH > 1 ? mipH / 2 : 1;
+ VkImageMemoryBarrier bd{ VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER };
+ bd.image = cube_image; bd.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, m, 1, 0, 6 };
+ bd.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; bd.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
+ bd.srcAccessMask = 0; bd.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
+ vkCmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &bd);
+
+ VkImageBlit blit{};
+ blit.srcOffsets[1] = { mipW, mipH, 1 }; blit.srcSubresource = { VK_IMAGE_ASPECT_COLOR_BIT, m - 1, 0, 6 };
+ blit.dstOffsets[1] = { nW, nH, 1 }; blit.dstSubresource = { VK_IMAGE_ASPECT_COLOR_BIT, m, 0, 6 };
+ vkCmdBlitImage(cmd, cube_image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
+ cube_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &blit, VK_FILTER_LINEAR);
+
+ VkImageMemoryBarrier bs{ VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER };
+ bs.image = cube_image; bs.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, m, 1, 0, 6 };
+ bs.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; bs.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
+ bs.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; bs.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
+ vkCmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &bs);
+ mipW = nW; mipH = nH;
+ }
+ VkImageMemoryBarrier fin{ VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER };
+ fin.image = cube_image; fin.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, CUBE_MIPS, 0, 6 };
+ fin.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; fin.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
+ fin.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; fin.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
+ vkCmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0, 0, nullptr, 0, nullptr, 1, &fin);
+ }
+
VK_CHECK(vkEndCommandBuffer(cmd));
VkSubmitInfo si{ VK_STRUCTURE_TYPE_SUBMIT_INFO }; si.commandBufferCount = 1; si.pCommandBuffers = &cmd;
VK_CHECK(vkQueueSubmit(graphics_queue, 1, &si, VK_NULL_HANDLE));