Resolve z fighting (#134)

* Fix already downloaded toolchain re-use

the following mv command would move $work into $toolchain_dir if the $toolchain_dir folder already existed.

* resolve z-fighting
This commit is contained in:
theofficialgman
2026-09-09 08:47:14 -04:00
committed by GitHub
parent 407f8a7190
commit 452b478bb3
8 changed files with 81 additions and 25 deletions
+1
View File
@@ -233,5 +233,6 @@ EOF
rm -rf "$test_dir"
trap - EXIT
rm -rf "$toolchain_dir"
mv "$work" "$toolchain_dir"
echo "prepare-portable-tools.sh: toolchain ready at $toolchain_dir ($(du -sh "$toolchain_dir" | cut -f1))"
+22 -6
View File
@@ -168,7 +168,12 @@ struct RenderPass {
Range resolveUniformRange;
std::array<u32, 3> resolveCopyFilterCoefficients{0, 64, 0};
Vec4<float> clearColorValue{0.f, 0.f, 0.f, 0.f};
float clearDepthValue = 1.f;
// 1.f is the forward-Z "farthest" clear value; under UseReversedZ farthest is 0.f instead (see
// gx::clear_depth_value(), which the main render pass explicitly overrides this default with -
// any OTHER pass that keeps this default, e.g. an offscreen render-to-texture pass composited
// later, needs the same reversed-Z-aware value or its depth buffer starts "already nearest",
// failing every subsequent depth test and making whatever's drawn into it vanish).
float clearDepthValue = gx::UseReversedZ ? 0.f : 1.f;
CommandList commands;
bool clearColor = true;
bool clearDepth = true;
@@ -757,7 +762,9 @@ void begin_offscreen(uint32_t width, uint32_t height) {
.targetSize = {width, height, 1},
.msaaSamples = 1,
.clearColorValue = {0.f, 0.f, 0.f, 0.f},
.clearDepthValue = 1.f,
// See the RenderPass::clearDepthValue default's comment: this offscreen pass gets its own
// depth buffer, and the farthest clear value is 0.f, not 1.f, under UseReversedZ.
.clearDepthValue = gx::UseReversedZ ? 0.f : 1.f,
.clearColor = true,
.clearDepth = true,
};
@@ -1410,10 +1417,19 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
switch (cmd.type) {
case CommandType::SetViewport: {
const auto& vp = cmd.data.setViewport;
// WebGPU requires 0 <= minDepth <= maxDepth <= 1, and the guest's (near, far) order is already
// reproduced in clip space. Passing the raw swapped pair diverged per backend in release builds.
const float minDepth = std::clamp(std::min(vp.znear, vp.zfar), 0.0f, 1.0f);
const float maxDepth = std::clamp(std::max(vp.znear, vp.zfar), 0.0f, 1.0f);
// WebGPU requires 0 <= minDepth <= maxDepth <= 1. vp.znear/vp.zfar are in GX's own distance
// terms (0 = near); under UseReversedZ the host depth-buffer storage direction is flipped
// (near = 1, far = 0), so this range has to be remapped through 1-x the same way the
// projection matrix, depth compare function, and clear value all are - a plain min/max clamp
// (the previous code here) maps a *restricted* range (e.g. a viewport deliberately narrowed
// to force something to draw "in front of everything") to the wrong end of the buffer: what
// should land near the near-storage-extreme (1.0) instead lands near the far-storage-extreme
// (0.0), so anything else drawn afterward at its true depth wins the compare test and the
// "in front" geometry silently vanishes. A full [0,1] viewport is unaffected either way,
// which is why this only broke specific elements, not the whole scene. Matches upstream
// aurora's apply_viewport (lib/gfx/encoding.cpp) exactly.
const float minDepth = gx::UseReversedZ ? 1.0f - vp.zfar : vp.znear;
const float maxDepth = gx::UseReversedZ ? 1.0f - vp.znear : vp.zfar;
pass.SetViewport(vp.left, vp.top, vp.width, vp.height, minDepth, maxDepth);
} break;
case CommandType::SetScissor: {
+1 -1
View File
@@ -92,7 +92,7 @@ struct Params {
constexpr std::string_view ReversedZBody = R"(
fn gx_z24(depth: f32) -> u32 {
return min(u32(clamp(depth, 0.0, 1.0) * 16777216.0), 0x00ffffffu);
return min(u32(clamp(1.0 - depth, 0.0, 1.0) * 16777215.0 + 0.5), 0x00ffffffu);
}
)"sv;
+1 -1
View File
@@ -137,7 +137,7 @@ fn gx_z24_at_coord(unclamped_coord: vec2i) -> u32 {
let tex_size = vec2i(textureDimensions(src));
let coord = clamp(unclamped_coord, vec2i(0), tex_size - vec2i(1));
let depth = textureLoad(src, coord, 0);
return min(u32(clamp(depth, 0.0, 1.0) * 16777216.0), 0x00ffffffu);
return min(u32(clamp(1.0 - depth, 0.0, 1.0) * 16777215.0 + 0.5), 0x00ffffffu);
}
)"s
: R"(
+13 -4
View File
@@ -1416,23 +1416,32 @@ static inline GXBlendFactor remove_dst_alpha_usage(GXBlendFactor fac) {
}
}
// GX_LEQUAL etc. describe "pass if this pixel is closer than/equal to what's stored" in GX's own
// distance terms, independent of how that distance is encoded as a host depth value. Under
// UseReversedZ the encoding is flipped (near=1, far=0), so "closer" now corresponds to a *larger*
// stored value, not a smaller one - the ordered compare functions (LESS/LEQUAL/GREATER/GEQUAL)
// must invert to match, or the depth test silently runs backwards (verified directly: this was
// the actual cause of a bug report after the projection/shader half of the reverse-Z fix
// eliminated the double-negation that used to accidentally keep the unreversed comparisons
// correct - LEQUAL now needs GreaterEqual, not LessEqual, once the encoding it's testing against
// is genuinely reversed). Matches upstream aurora's to_compare_function exactly.
static inline wgpu::CompareFunction to_compare_function(GXCompare func) {
switch (func) {
DEFAULT_FATAL("invalid depth fn {}", underlying(func));
case GX_NEVER:
return wgpu::CompareFunction::Never;
case GX_LESS:
return wgpu::CompareFunction::Less;
return UseReversedZ ? wgpu::CompareFunction::Greater : wgpu::CompareFunction::Less;
case GX_EQUAL:
return wgpu::CompareFunction::Equal;
case GX_LEQUAL:
return wgpu::CompareFunction::LessEqual;
return UseReversedZ ? wgpu::CompareFunction::GreaterEqual : wgpu::CompareFunction::LessEqual;
case GX_GREATER:
return wgpu::CompareFunction::Greater;
return UseReversedZ ? wgpu::CompareFunction::Less : wgpu::CompareFunction::Greater;
case GX_NEQUAL:
return wgpu::CompareFunction::NotEqual;
case GX_GEQUAL:
return wgpu::CompareFunction::GreaterEqual;
return UseReversedZ ? wgpu::CompareFunction::LessEqual : wgpu::CompareFunction::GreaterEqual;
case GX_ALWAYS:
return wgpu::CompareFunction::Always;
}
+8 -1
View File
@@ -485,7 +485,14 @@ const gfx::TextureBind& get_texture(GXTexMapID id) noexcept;
void resolve_sampled_textures(const ShaderInfo& info) noexcept;
inline float clear_depth_value() {
return std::min(static_cast<float>(g_gxState.clearDepth) / 16777216.f, 16777215.f / 16777216.f);
// g_gxState.clearDepth is in GX's own distance terms (0 = near, larger = farther), independent of
// how UseReversedZ encodes that as a host depth value - it must be re-mapped the same way the
// projection matrix and depth compare function are, or the buffer clears to the wrong extreme
// (verified directly: matches upstream aurora's clear_depth_value, which does this same inversion
// and was the second missing piece alongside to_compare_function's compare-op inversion).
const float normalizedDepth =
std::min(static_cast<float>(g_gxState.clearDepth) / 16777216.f, 16777215.f / 16777216.f);
return UseReversedZ ? (1.f - normalizedDepth) : normalizedDepth;
}
inline bool render_target_has_alpha(GXPixelFmt pixelFmt) noexcept { return pixelFmt == GX_PF_RGBA6_Z24; }
+23 -8
View File
@@ -993,11 +993,13 @@ wgpu::ShaderModule build_shader(const ShaderConfig& config) noexcept {
"\n let clip_base = select(clip_a, clip_b, use_b);"
"\n out.pos = vec4f(clip_base.xy + offset_ndc * clip_base.w, clip_base.zw);";
}
if constexpr (UseReversedZ) {
vtxXfrAttrsPre += "\n out.pos.z = -out.pos.z;";
} else {
vtxXfrAttrsPre += "\n out.pos.z += out.pos.w;";
}
// The near/far depth correction used to be applied here per-vertex (out.pos.z = -out.pos.z for
// reversed, or += out.pos.w for forward), redundantly on top of the same correction already
// folded into ubuf.proj by effective_projection() (shader_info.cpp) - applying it twice canceled
// out for the common case (any draw where effective_projection() decides to flip), silently
// making "reversed" Z behave identically to forward Z. It is now applied exactly once, in the
// projection matrix alone (matching upstream aurora commit 1dde08fa: "Move depth correction to
// projection matrix"), so nothing needs to happen to out.pos.z here.
// GX rasterizes at a 7/12 pixel center when antialiasing is disabled, while WebGPU rasterizes at 1/2.
vtxXfrAttrsPre +=
"\n let gx_pixel_center_correction = "
@@ -1465,7 +1467,14 @@ wgpu::ShaderModule build_shader(const ShaderConfig& config) noexcept {
textureDependency.texMapId, uvIn);
}
std::string fogDepthExpr = UseReversedZ ? "in.pos.z" : "(1.0 - in.pos.z)";
// in.pos.z is the host NDC z (forward: 0=near/1=far; reversed: 1=near/0=far post-fix), but this
// expression needs to produce GX's own native distance term (always 0=near/1=far, matching how
// g_gxState.clearDepth/clear_depth_value() are interpreted before their own UseReversedZ
// inversion) - forward already matches directly; reversed needs the same 1-x flip everything
// else reversed-Z-aware uses. This was backwards (verified directly against upstream aurora's
// identical expression in build_shader_source), which fed both fog density and the GX_ZT_ADD
// z-texture path the wrong distance value.
std::string fogDepthExpr = UseReversedZ ? "(1.0 - in.pos.z)" : "in.pos.z";
std::string fogZCoordExpr =
fmt::format("u32(round(clamp({}, 0.0, 1.0) * 16777216.0))", fogDepthExpr);
if (usesZTextureDepth) {
@@ -1498,7 +1507,7 @@ wgpu::ShaderModule build_shader(const ShaderConfig& config) noexcept {
fragmentFn += fmt::format(
"\n let oldZ = u32(round(clamp({0}, 0.0, 1.0) * 16777216.0));"
"\n ztexCoord = (ztexCoord + oldZ) & 0x00ffffffu;",
UseReversedZ ? "in.pos.z" : "(1.0 - in.pos.z)");
UseReversedZ ? "(1.0 - in.pos.z)" : "in.pos.z");
}
fragmentFn += "\n let ztexDepth = f32(ztexCoord) / 16777216.0;";
fogZCoordExpr = "ztexCoord";
@@ -1639,7 +1648,13 @@ wgpu::ShaderModule build_shader(const ShaderConfig& config) noexcept {
" @builtin(frag_depth) depth: f32,\n"
"};";
fragmentFn += fmt::format("\n let fragDepth = {}ztexDepth;", UseReversedZ ? "" : "1.0 - ");
// ztexDepth is in GX's native distance terms (0=near/1=far, see fogDepthExpr's comment above),
// but frag_depth must be written in the same host NDC-z convention in.pos.z itself uses -
// forward matches directly (no change), reversed needs the same 1-x flip. This was backwards
// the same way fogDepthExpr was (verified by the same derivation, since aurora upstream has no
// directly equivalent line here to cross-check against - this z-texture-depth-output path
// appears to be specific to this fork).
fragmentFn += fmt::format("\n let fragDepth = {}ztexDepth;", UseReversedZ ? "1.0 - " : "");
fragmentReturnType = "FragmentOutput";
fragmentReturn =
" var out: FragmentOutput;\n"
+12 -4
View File
@@ -548,14 +548,22 @@ constexpr size_t kStagedUniformBytes =
96 + sizeof(Mat4x4<float>) + sizeof(Mat3x4<float>) * (MaxPostexMtx + MaxPnMtx);
// The host viewport always receives the normalized GX depth window (render_pass_impl clamps to minDepth <= maxDepth).
//
// Folds the near/far depth correction the vertex shader used to apply per-vertex directly into the
// projection matrix instead (matching upstream aurora commit 1dde08fa, "Move depth correction to
// projection matrix") - valid because the correction is a linear combination of the z/w rows, so
// applying it once here to the row is equivalent to applying it once per-vertex to the dot product,
// and it must be applied exactly once: doing it here AND in the shader (the previous bug) canceled
// the negation out for `flip`, silently making "reversed" Z behave identically to forward Z.
// `flip` decides which of the two single-application forms this draw needs: true bakes in the
// reversed-Z inversion (z' = -z), false bakes in the forward-Z near/far combination (z' = z + w) -
// exactly one always applies, never both, and never neither.
static Mat4x4<float> effective_projection() noexcept {
const auto& vp = g_gxState.renderViewport;
const bool flip = (vp.znear <= vp.zfar) == UseReversedZ;
Mat4x4<float> proj = g_gxState.proj;
if (flip) {
for (size_t i = 0; i < 4; ++i) {
proj.m2.m[i] = -(proj.m2.m[i] + proj.m3.m[i]);
}
for (size_t i = 0; i < 4; ++i) {
proj.m2.m[i] = flip ? -proj.m2.m[i] : (proj.m2.m[i] + proj.m3.m[i]);
}
return proj;
}