diff --git a/Benchmarks/RENDERING_OPTIMIZATIONS.md b/Benchmarks/RENDERING_OPTIMIZATIONS.md new file mode 100644 index 0000000000..171a730b5d --- /dev/null +++ b/Benchmarks/RENDERING_OPTIMIZATIONS.md @@ -0,0 +1,314 @@ +# Rendering optimizations + +Living notes for work on `optimization/rendering-improvements`. Each entry is what changed, why, and what can break. + +Measure with **`ReleaseTracyProfiler`**, Shift+F10, Main window GPU timestamps (or RenderDoc on that build). Plain Release `rs_stats` will not show these GPU copies. + +Single-frame Tracy CSVs are noisy (~1–2 ms Frame swing). Compare named passes (HUD, sun, sorted, `combine_1` must match) before trusting Frame. + +--- + +## Skip unused depth snapshots + +**Status:** done +**Files:** `ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp` + +### What it did + +Every frame used to `CopyResource` the scene depth buffer three times: + +| Destination | When it is taken | Who reads it | +|---|---|---| +| `rt_tempzb` (`$user$temp_zb`) | After world geo, **before** HUD | 3D-scope z-write (`3dss_zwrite.ps`) | +| `rt_tempzb_dof` (`$user$zbuffer_dof`) | After HUD, **before** scope depth | DOF (`ogsr_dof.s`) | +| `rt_zbuffer` (`$user$zbuffer`) | After HUD + scope depth | TAA, DLSS, FSR3, puddles | + +The first two copies are now skipped unless that feature will run this frame: + +- `rt_tempzb` only if `mapScopeHUD` is not empty (same map `r_dsgraph_render_hud_scope_depth` draws). +- `rt_tempzb_dof` only if DOF params are non-zero (same test as `phase_dof`). + +`rt_zbuffer` still copies every frame. Scene depth stays bound as a DSV, so later passes cannot sample it in place. + +Copies now use `rt_Base_Depth->pSurface` instead of `GetResource` / `Release` on the DSV each time. + +### What it can affect + +- **3D scopes:** the tube should still show the world without HUD weapons in the depth. If `mapScopeHUD` is empty while a scope still samples `$user$temp_zb`, depth in the tube will be stale or wrong. +- **DOF (zoom / reload / script `set_dof_params`):** weapon and world should still defocus using HUD-inclusive depth. If DOF runs while all four `dof_params` look zero at copy time, it will sample a stale `$user$zbuffer_dof`. +- **TAA / DLSS / FSR / puddles:** unchanged; they still get a fresh `$user$zbuffer`. +- **Tracy markers:** `copy_zbuffer_scope` and `copy_zbuffer_scope_depth` only appear when the copy actually happens. `copy_zbuffer` should always appear. + +### Sticky `copy_zbuffer_scope_depth` after first aim + +**Status:** fixed +**Files:** `ogsr_engine/xrGame/Weapon.cpp` + +`CWeapon::UpdateDof` used to `set_dof_params` and *then* `clamp` the fade to `[0, 1]`. Fade-out only runs while `dof_zoom_effect > 0`, so the last frame wrote a leftover (tiny or negative) and never wrote zeros. `phase_dof` and the depth-copy gate both use `fis_zero` on those four values, so after the first ADS they kept running until restart. Clamp now happens first; the last fade frame exports `(0,0,0,0)`. + +`copy_zbuffer_scope` is independent (3D-scope HUD map). That one should already vanish when you unaim. + +### How to check + +- Walk around, no scope, no DOF: two extra depth copies gone; image unchanged. +- 3D scope: reticle / world depth in the tube looks like before (no HUD mesh in the glass). +- Iron sights / reload DOF: blur still respects nearby HUD and world depth. +- DLSS/FSR/TAA: no extra ghosting vs before. `copy_zbuffer` still fires. + +--- + +## Tracy editor: GPU pass filter and CSV export + +**Status:** done (Tracy build only, not a rendering change) +**Files:** `ogsr_engine/xrGame/embedded_editor/embedded_editor_main.cpp` + +Shift+F10 Main window: case-insensitive name filter, scrollable pass list, Export CSV of the current frame (all depths) to `$app_data_root$/gpu_passes_YYYYMMDD_HHMMSS.csv` and the clipboard (`index,stack,name,time_ms`). Depth slider still applies to the on-screen list only. Needs `Build_ReleaseTracy.cmd` / `CONFIGURATION_GA=ReleaseTracyProfiler`. Does not affect Release visuals or GPU work. + +--- + +## Post-process ping-pong: CAS after DLSS/FSR + +**Status:** done (slice 1; slice 2 below removes the rest) +**Files:** `RenderTargetPhaseAA.cpp`, `r4_rendertarget.h`, `r4_rendertarget_phase_combine.cpp`, `contrast_adaptive_sharpening.s` + +### What it did + +Display-sized post used to draw into `rt_Generic_combine` then `CopyResource` into `rt_Postprocess_0` because shaders sample `$user$postprocess0`. With DLSS/FSR that was two full RGBA16F copies before anything else: upscale output → postprocess0, then CAS → copy back. + +When CAS runs after a temporal upscaler: + +- Skip the upscale → postprocess0 copy +- CAS samples `$user$generic_combine` (element 1) and writes `rt_Postprocess_0` +- Skip the CAS copy + +TAA/SMAA/no-AA keep the old seed (`copy_pp_from_generic0`, CAS element 0). Do **not** ping-pong with `rt_Generic_0` — that buffer is render-sized when DLSS/FSR is on. + +### What it can affect + +- **CAS after DLSS/FSR:** sharpening should match. If shaders did not recook, element 1 is missing and the engine falls back to the old copies. +- **3D scopes:** see slice 2. `copy_pp_3dss` is no longer “whenever a scope is up”; it only runs when the latest image is still in postprocess0. + +### How to check + +- Tracy, DLSS+CAS, no ADS: `copy_pp_after_upscale` and `copy_pp_cas` gone; `CAS` is draw-only. +- TAA or `r_aa_cas 0`: `copy_pp_after_upscale` / `copy_pp_from_generic0` still seed the pair. + +### Measured (cordon_pingpong.csv) + +Versus `cordon_depth_optimization.csv`: CAS 0.078 → 0.064 ms. Leftover `copy_pp_ssss` / `copy_pp_combine2` were 0.026 ms each. HUD/sorted did not match that capture — do not use its Frame. + +--- + +## Post-process ping-pong: remaining display-sized copies + +**Status:** done +**Files:** `r4_rendertarget.h`, `RenderTargetRenderScreenQuad.cpp`, `RenderTargetPhaseAA.cpp`, `RenderTargetPhaseSSSS.cpp`, `r4_rendertarget_phase_combine.cpp`, `r4_rendertarget_phase_PP.cpp`, `r4_rendertarget_phase_lut.cpp`, `rendertarget_phase_dof.cpp`, `rendertarget_phase_gasmask_dudv.cpp`, `rendertarget_phase_nightvision.cpp`, `rendertarget_phase_thermalvision.cpp`, `RenderTargetPhaseRainDrops.cpp`, `rendertarget_phase_fakescope.cpp` + +### What it did + +Each post writer used to draw into combine then `CopyResource` into postprocess0 (~0.026 ms per blit). The two display-sized buffers are now a ping-pong pair: + +- `m_pp_current_is_combine` tracks which one holds the latest image (`false` = `rt_Postprocess_0`). +- After `set_Element`, `pp_remap_scene_srv` rebinds any `$user$postprocess0` / `$user$generic_combine` slot to the current buffer via the existing `CTexture` SRV (no `surface_set`). +- Each writer samples current, draws into the other RT, then flips. Remap is on only between `BeginPostprocess` and `phase_pp` so TAA/SMAA/SSR are untouched. + +DLSS/FSR still write combine. Skipping the upscale→postprocess0 copy marks combine as current so CAS element 1 does not sample the RT it is about to write. + +3DSS still samples `$user$generic_combine` while drawing into `rt_Postprocess_0` (cannot remap: that would bind the color RT as an SRV). If SSSS or heat overlay already left the latest image in combine, that copy is skipped. If CAS (or another ping-pong write) left it in postprocess0, `copy_pp_3dss` still syncs combine first. + +`copy_pp_after_upscale` / `copy_pp_from_generic0` remain when CAS is off or there is no temporal upscaler — seed into the pair, not a per-pass tax. + +### What it can affect / regressions to watch + +Two failure modes: **stale buffer** (effect missing, or last frame’s effect) and **3D-scope glass sampling the wrong RT**. + +**3D scopes** (reticle is not remapped): + +| Setup | If broken, the tube shows | +|---|---| +| DLSS/FSR + CAS, shafts **off** | Unsharpened world (`copy_pp_3dss` should still fire) | +| Shafts **on** | World without shafts (copy should *not* fire; combine already has them) | +| Thermal NVG (`pnv` 2/3) | Glass missing the heat overlay, or reticle heat-tinted | +| Aim / unaim / swap weapons | One frame of garbage, or it sticks | +| `r_dlss_3dss_scale_factor` > 1 | Separate `$user$generic_combine_scope` path; should be unchanged | + +**Missed remap** (odd vs even flip count). `phase_pp` samples combine; after an odd number of flips the latest image is in postprocess0 and only remap saves it. Drunk / radiation / gray / dual-vision / colormap is the cheapest way to see that. Flip the count on purpose: CAS only; CAS + shafts; add iron-sight DOF, LUT, gasmask (two flips), NVG 1, rain. + +If a pass sampled the wrong buffer: + +- **Shafts:** rays missing, or stuck from the previous camera angle +- **DOF:** HUD in focus while the world isn’t, or the reverse after unaim (also still check the sticky-DOF fix above) +- **LUT:** grade pops in a frame late +- **Gasmask:** visor reflection / breath on an unmasked scene +- **NVG 1:** green on a pre-DOF or pre-mask frame +- **Rain drops:** drops on a scene without shafts/DOF +- **CAS:** no sharpen, or a D3D11 “simultaneous SRV and RT” error if it samples the RT it writes +- **`combine_2` motion blur:** blur from the previous pose when turning quickly + +**Fallback seeds** (one look each): TAA, SMAA (skips CAS), `r_aa_cas 0`, FSR, AA off. You should still see `copy_pp_after_upscale` or `copy_pp_from_generic0`, and no per-pass `copy_pp_*` after that. + +Recook shaders if CAS looks off. Alt-tab / change DLSS quality / change resolution once (RTs recreate; flags reset next frame). + +### How to check + +- Tracy, DLSS+CAS, shafts on, no ADS: `copy_pp_ssss`, `copy_pp_combine2`, `copy_pp_dof`, `copy_pp_lut`, `copy_pp_nvg`, `copy_pp_gasmask_*`, `copy_pp_rain` gone. Image unchanged. +- 3D scope, shafts off: `copy_pp_3dss` present; tube is sharpened. +- 3D scope, shafts on: `copy_pp_3dss` absent; tube has shafts + CAS. +- TAA or `r_aa_cas 0`: seed copy still fires; later per-pass copies still gone. + +### Measured (cordon_pingpong2.csv) + +Versus slice 1 (`cordon_pingpong.csv`): `copy_pp_ssss` / `copy_pp_combine2` gone. `phase_ss_ss` 0.239 → 0.213 ms (−0.026, matches the 0.0261 ms copy). `combine_2` 0.248 → 0.220 ms (−0.028, matches 0.0264). CAS stays draw-only at 0.063 ms. Sun / lights / DLSS / blur / bloom match. ~0.05 ms this slice, ~0.07 ms across both ping-pong slices. Frame 60.8 → 59.2 is mostly a cheaper `combine_1` plus unmatched sorted transparents — do not quote that as ping-pong. HUD finally matched baseline (0.10 vs 0.09 ms). + +--- + +## Gate whole PHASE_BLUR + +**Status:** skipped +SSFX bloom is forced on and samples `$user$blur_2`. The six-pass pyramid (~0.16 ms) always has a consumer. A 1/4+1/8 skip would be tiny; not worth it. + +--- + +## Skip SSSS RT clears when shafts off + +**Status:** skipped +Two display-sized clears. The flare comment is stale (`combine_2_naa.ps` does not sample `s_mask_flare_*`). Tiny; left alone. + +--- + +## SSFX bloom draws emissive twice + +**Status:** skipped +**Files:** `ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp` (472–479) + +With `R2FLAG_SSFX_BLOOM` (always on) the same `mapEmissive` / `mapHUDEmissive` meshes are drawn into `rt_Accumulator` (lighting) and again into render-sized `rt_ssfx_bloom_emissive` (RGBA8, sampled by `ogsr_bloom.ps` without the scene threshold). Plus a full-res clear of that RT every frame. + +### Why it is not worth it + +- **Empty view (Cordon captures):** `DEFER_SELF_ILLUM` is already **0.000 ms**. Empty maps skip both draws. The leftover is one RGBA8 clear + RT bind — same class as the SSSS clears we skipped. Not in the same league as the 0.026 ms display-sized RGBA16F copies. +- **Lamps / PDA / detectors on screen:** you pay a second geometry pass of those meshes, not a fullscreen blit. A handful of quads, not perceptible bandwidth. +- **MRT to draw once into both RTs** needs every emissive pixel shader (`accum_emissive*`, `accum_lamp`, addon `l_special`) to write `SV_Target0` + `SV_Target1` and recook. Wrong output = missing lamp lighting or missing lamp bloom. Not worth that risk for a sub-copy saving. + +Skipping the clear when the maps are empty still ghosts last frame’s lamps unless bloom is told to ignore `s_emissive` — another shader recook for the same tiny clear. + +Old `phase_bloom` (~0.020 ms, includes luminance for exposure) still runs and `combine_2_naa.ps` still samples `$user$bloom1`. That is a separate dual-bloom question; do not rip it out without a visual check. `phase_ssfx_bloom` itself (~0.088 ms) is the real bloom work. + +--- + +## Remaining + +- Full-res SSR when wet +- `WaitOnSwapChain` busy-wait + +--- + +## Dedicated AO pass and optional half-resolution evaluation + +**Status:** implemented; in-game visual and performance comparison pending + +**Files:** `r4_rendertarget_phase_ao.cpp`, `r4_rendertarget_phase_combine.cpp`, `r4_rendertarget.{h,cpp}`, `xrRender_console.{h,cpp}`, `combine.s`, `combine_1.ps`, `combine_1_ao.ps`, `ogsr_ao*.{ps,s}` + +### What changed + +SSDO and GTAO can now produce a separate visibility texture, consumed by `combine_1` before the existing colored-AO and ambient-light treatment. This creates the common output used by the subsequent XeGTAO integration described below. + +`r_ao_resolution` switches at runtime: + +| Value | Behavior | GPU markers | +|---|---|---| +| `legacy` | Original AO inside `combine_1`; no dedicated AO draws | `combine_1` | +| `full` (default) | Existing AO at full **internal render** resolution into R16F | `phase_ao` / `ao_evaluate_full` | +| `half` | One AO evaluation per 2x2 render-pixel block, then depth/normal-aware reconstruction into R16F | `phase_ao` / `ao_evaluate_half`, `ao_resolve` | + +AO method and sample quality still use `r_ao_mode` and `r2_ssao`. These are shader permutations, so apply changes with `vid_restart` as before. With AO disabled at shader creation, the new targets and AO shaders are not allocated and no AO pass runs. + +Full mode is a comparison baseline and integration step, **not a promised speedup**: it adds a texture write/read without reducing the sample count. Half mode provides the work reduction, at the cost of spatial detail. With DLSS/FSR, both sizes are relative to the already reduced internal render resolution, not display resolution. + +Half mode chooses the nearest covered G-buffer pixel in each 2x2 block and stores visibility, view depth and the packed normal in RGBA16F. A four-tap resolve rejects mismatched depth/normals. If no compatible sample exists, it returns unoccluded visibility rather than borrowing foreground AO. Odd dimensions use rounded-up half targets and clamped integer coordinates. Full mode preserves the original interpolated UVs and jitter reconstruction, important for SSDO's noise hash. + +The final AO target is fully overwritten each frame. There is no new temporal history, no copy-back, and no additional depth snapshot. The combine viewport is restored after AO. Target recreation follows the existing renderer reset path. + +### How to check + +1. Build the engine and deploy the changed and new shader files together. Ensure cached shaders are recompiled. Enable either AO method and a nonzero `r2_ssao`, then `vid_restart`. +2. Fixed camera/weather/settings: switch between `r_ao_resolution legacy` and `r_ao_resolution full`. Lighting should match apart from R16F visibility rounding. Full-mode time must include **both** `phase_ao` and `combine_1`. +3. Switch to `r_ao_resolution half`. Compare the sum of those passes and total frame time over multiple frames, not a single capture. Do not count `phase_ao` and its children twice. +4. Inspect grass, fence wires, building corners, foreground silhouettes and HUD weapons. Half mode can lose thin-surface AO or show bright gaps where no compatible background sample exists; use full mode when that tradeoff is unacceptable. +5. Check scope aim/unaim, TAA, SMAA, AA off, DLSS and FSR, including odd internal dimensions and resolution/quality changes. Look for viewport clipping, unstable contact shadows and shader resource binding warnings. +6. Disable AO and restart the renderer: no `phase_ao` marker; lighting should match the old AO-off path. + +### Validation + +`Benchmarks/validate_ao_shaders.ps1` compiles 33 shader variants with FXC: both methods, all quality levels including off, legacy/buffer composition, full/half evaluation, and reconstruction. This is compilation coverage, not an in-game performance measurement. + +The full ReleaseTracyProfiler engine build and executable link passed (`bin_x64/xrEngine.exe`), with existing compiler warnings. A standalone D3D11 WARP smoke test passed 14 cases: full-resolution visibility matched the legacy calculation for SSDO and GTAO on even, odd and 1x1 targets, with and without jitter; half-resolution results were finite and sky pixels remained unoccluded; reconstruction rejected both depth and normal discontinuities. In-game screenshots and timings are still required before claiming a speedup or visual equivalence in actual scenes. + +--- + +## Selectable XeGTAO + +**Status:** implemented and working in-game; performance and visual comparison remain user-measured + +**Files:** `r4_rendertarget_phase_xegtao.cpp`, AO routing/target lifecycle and console files, `dx10ShaderResourceStateCache.cpp`, `ogsr_xegtao*`, `shaders/r3/xegtao/`. + +### What changed + +`r_ao_mode st_xegtao` selects the MIT-licensed Intel implementation, pinned to upstream commit `a5b1686c7ea37788eeb3576b5be47f7c03db532c`. The default method remains SSDO. Existing SSDO/GTAO full/half/legacy modes are retained. + +The XeGTAO path always uses **full internal render resolution**. `r_ao_resolution` applies only to SSDO/GTAO; `half` and `legacy` do not change XeGTAO's resolution or send it through legacy inline AO. The initialization log makes this explicit. With DLSS/FSR, internal resolution is already smaller than display resolution. + +1. An adapted compute prefilter reads existing linear view depth and builds five R16F depth mips in one dispatch. It does not copy hardware depth or regenerate normals. +2. XeGTAO evaluates visibility using the G-buffer's octahedral normals. `r2_ssao st_opt_low`, `st_opt_medium`, and `st_opt_high` select 1x2, 2x2, and 3x3 slice/step presets respectively (samples are taken on both sides of each slice). +3. One upstream edge-aware denoise pass filters R8 visibility and packed R8 edges. +4. A full-screen export converts integer visibility into the existing R16F AO target. `combine_1_ao` retains the existing colored-AO and ambient-light treatment. Sky exports exactly 1. + +The compute textures are padded to multiples of 16, with border replication during prefiltering. Projection constants compensate for padding and match OGSR's existing jittered view-space reconstruction. Math is FP32 on SM5; compact depth/AO storage does not require native FP16 arithmetic or a newer shader compiler. There is no separate AO temporal history: noise advances with TAA/DLSS/FSR, otherwise it stays fixed. + +Resources and shaders are allocated only when XeGTAO and nonzero AO quality are active at target creation; they are released on renderer reset. The legacy half-size target is not allocated for XeGTAO. Direct compute uses explicit UAV/SRV transitions and clears/invalidate state before returning to graphics. The backend's reset now clears compute SRV cache entries and dirty ranges too, as it already did for other stages. + +`r_xegtao_radius` adjusts the view-space radius live (default 0.5, range 0.05-4). Radius/art direction is not calibrated to match legacy GTAO, so compare quality and cost at acceptable visual settings, not just matching preset names. + +### How to check + +Build the engine and deploy all new shader files, **including the `xegtao` subdirectory**. Then: + +```text +r_ao_mode st_xegtao +r2_ssao st_opt_medium +r_xegtao_radius 0.5 +vid_restart +``` + +- Optional shader compilation coverage: run `Benchmarks/validate_xegtao_shaders.ps1` (15 SM5 variants). +- Compare a fixed scene against `st_ssdo` and `st_gtao`, restarting after each method/quality change. Include AO off as a reference. Test the existing optimized half-resolution mode separately; XeGTAO is not guaranteed to beat half-resolution AO. +- Capture `phase_ao` and its `ao_xegtao` children: `ao_xegtao_prefilter`, `ao_xegtao_evaluate`, `ao_xegtao_denoise`, `ao_xegtao_export`. Compare total frame time and `phase_ao + combine_1`; do not sum parents and their children. The outer scope also captures state-transition overhead. +- Check flat walls, corners, thin foliage/fences, depth discontinuities, screen borders, sky silhouettes, and HUD weapons/aimed scopes. Inspect noise and dark halos both stationary and in motion. Weapon projection/depth conventions are inherited from the G-buffer and need particular visual scrutiny. +- Check AA off/SMAA and TAA/DLSS/FSR, multiple internal sizes (including odd sizes), repeated `vid_restart`, and AO off. AO off must allocate no XeGTAO resources and emit no AO passes. +- Watch for shader errors, device/debug-layer binding warnings, invalid pixels, or changed unrelated rendering. Performance and image-quality claims should be based on these measurements. + +### TODO + +- Add half-resolution XeGTAO evaluation and an AO-aware resolve path. +- Add optional bent-normal output for directional ambient lighting. +- Add dedicated temporal accumulation/history if testing shows a benefit over the existing TAA/DLSS/FSR noise progression. +- Add dedicated weapon/HUD handling if G-buffer depth conventions produce visible mismatches. +- Benchmark XeGTAO against full- and half-resolution SSDO/GTAO and tune the default radius/quality presets. + +Static checks passed for project/filter XML and validation-script PowerShell syntax. A 54-case arithmetic sanity check matched the padded XeGTAO reconstruction to the existing G-buffer formula at even, odd and tiny sizes, with and without jitter. These are not compilation or GPU execution tests. + +--- + +## Template for the next entry + +``` +## Short name + +**Status:** done | skipped +**Files:** … + +### What it did + +### What it can affect + +### How to check +``` diff --git a/Benchmarks/cordon_baseline.csv b/Benchmarks/cordon_baseline.csv new file mode 100644 index 0000000000..573ab51024 --- /dev/null +++ b/Benchmarks/cordon_baseline.csv @@ -0,0 +1,53 @@ +index;stack;name;time_ms +0;0;Frame;62.280 +1;1;CRender_Render;61.532 +2;2;SKY_RENDER;0.0430 +3;2;phase_scene_prepare;0.0072 +4;2;DEFER_PART0_SPLIT;0.3963 +5;3;dsgraph_render_static;0.3830 +6;3;dsgraph_render_dynamic;0.0133 +7;2;DEFER_TEST_LIGHT_VIS;0.0000 +8;2;DEFER_PART1_SPLIT;0.2194 +9;3;dsgraph_render_lods;0.0000 +10;3;DetailManager_Render;0.0819 +11;3;copy_zbuffer_scope;0.0229 +12;3;dsgraph_render_hud;0.0912 +13;3;copy_zbuffer_scope_depth;0.0211 +14;2;copy_zbuffer;0.0211 +15;2;DEFER_WALLMARKS;0.0040 +16;3;dsgraph_render_wmarks;0.0031 +17;2;DEFER_RAIN;0.0000 +18;2;DEFER_SUN;13.271 +19;3;render_sun_flush;13.271 +20;4;render_sun_cascade_0;0.6175 +21;4;render_sun_cascade_1;0.4690 +22;4;render_sun_cascade_2;0.2396 +23;3;accum_direct_blend;0.0000 +24;2;DEFER_SELF_ILLUM;0.0000 +25;2;DEFER_LIGHT;0.5745 +26;3;SHADOWED_LIGHTS;0.5100 +27;3;SPOT_LIGHTS_ACCUM_VOLUMETRIC;0.0604 +28;3;POINT_LIGHTS_ACCUM;0.0041 +29;2;DEFER_COMBINE;35.594 +30;3;CLOUDS_RENDER;0.0246 +31;3;combine_1;18.975 +32;3;render_forward;0.1155 +33;4;dsgraph_render_graph;0.0205 +34;5;dsgraph_render_static;0.0174 +35;5;dsgraph_render_dynamic;0.0031 +36;4;dsgraph_render_sorted;0.0942 +37;3;phase_combine_volumetric;0.0676 +38;3;render_distort_objects;0.0010 +39;4;dsgraph_render_distort;0.0000 +40;3;combine_distort;0.0648 +41;3;phase_bloom;0.0206 +42;3;combine_tonemap;0.0890 +43;3;DLSS;0.3947 +44;3;CAS;0.0775 +45;3;phase_ss_ss;0.2325 +46;4;phase_SS_SS_OGSE;0.2300 +47;3;PHASE_BLUR;0.1558 +48;3;phase_ssfx_bloom;0.0852 +49;3;combine_2;0.2482 +50;3;RenderFlares;0.0006 +51;3;phase_pp;0.0190 diff --git a/Benchmarks/cordon_compare.xlsx b/Benchmarks/cordon_compare.xlsx new file mode 100644 index 0000000000..fd55920797 Binary files /dev/null and b/Benchmarks/cordon_compare.xlsx differ diff --git a/Benchmarks/cordon_depth_optimization.csv b/Benchmarks/cordon_depth_optimization.csv new file mode 100644 index 0000000000..616af02d48 --- /dev/null +++ b/Benchmarks/cordon_depth_optimization.csv @@ -0,0 +1,51 @@ +index;stack;name;time_ms +0;0;Frame;64.133 +1;1;CRender_Render;63.365 +2;2;SKY_RENDER;0.0451 +3;2;phase_scene_prepare;0.0061 +4;2;DEFER_PART0_SPLIT;0.3994 +5;3;dsgraph_render_static;0.3840 +6;3;dsgraph_render_dynamic;0.0154 +7;2;DEFER_TEST_LIGHT_VIS;0.0000 +8;2;DEFER_PART1_SPLIT;0.4424 +9;3;dsgraph_render_lods;0.0010 +10;3;DetailManager_Render;0.0850 +11;3;dsgraph_render_hud;0.3553 +12;2;copy_zbuffer;0.0219 +13;2;DEFER_WALLMARKS;0.0041 +14;3;dsgraph_render_wmarks;0.0031 +15;2;DEFER_RAIN;0.0000 +16;2;DEFER_SUN;13.076 +17;3;render_sun_flush;13.076 +18;4;render_sun_cascade_0;0.6226 +19;4;render_sun_cascade_1;0.4444 +20;4;render_sun_cascade_2;0.2396 +21;3;accum_direct_blend;0.0000 +22;2;DEFER_SELF_ILLUM;0.0000 +23;2;DEFER_LIGHT;0.5704 +24;3;SHADOWED_LIGHTS;0.5048 +25;3;SPOT_LIGHTS_ACCUM_VOLUMETRIC;0.0614 +26;3;POINT_LIGHTS_ACCUM;0.0041 +27;2;DEFER_COMBINE;35.389 +28;3;CLOUDS_RENDER;0.0256 +29;3;combine_1;19.313 +30;3;render_forward;0.0227 +31;4;dsgraph_render_graph;0.0215 +32;5;dsgraph_render_static;0.0154 +33;5;dsgraph_render_dynamic;0.0061 +34;4;dsgraph_render_sorted;0.0000 +35;3;phase_combine_volumetric;0.0737 +36;3;render_distort_objects;0.0020 +37;4;dsgraph_render_distort;0.0010 +38;3;combine_distort;0.0726 +39;3;phase_bloom;0.0210 +40;3;combine_tonemap;0.0893 +41;3;DLSS;0.3995 +42;3;CAS;0.0778 +43;3;phase_ss_ss;0.2420 +44;4;phase_SS_SS_OGSE;0.2397 +45;3;PHASE_BLUR;0.1574 +46;3;phase_ssfx_bloom;0.0850 +47;3;combine_2;0.2473 +48;3;RenderFlares;0.0006 +49;3;phase_pp;0.0201 diff --git a/Benchmarks/cordon_pingpong.csv b/Benchmarks/cordon_pingpong.csv new file mode 100644 index 0000000000..f1d98e2a57 --- /dev/null +++ b/Benchmarks/cordon_pingpong.csv @@ -0,0 +1,53 @@ +index;stack;name;time_ms +0;0;Frame;60.795 +1;1;CRender_Render;60.037 +2;2;SKY_RENDER;0.0440 +3;2;phase_scene_prepare;0.0061 +4;2;DEFER_PART0_SPLIT;0.3973 +5;3;dsgraph_render_static;0.3809 +6;3;dsgraph_render_dynamic;0.0164 +7;2;DEFER_TEST_LIGHT_VIS;0.0000 +8;2;DEFER_PART1_SPLIT;0.1075 +9;3;dsgraph_render_lods;0.0020 +10;3;DetailManager_Render;0.0829 +11;3;dsgraph_render_hud;0.0225 +12;2;copy_zbuffer;0.0221 +13;2;DEFER_WALLMARKS;0.0049 +14;3;dsgraph_render_wmarks;0.0041 +15;2;DEFER_RAIN;0.0000 +16;2;DEFER_SUN;13.179 +17;3;render_sun_flush;13.179 +18;4;render_sun_cascade_0;0.6124 +19;4;render_sun_cascade_1;0.4690 +20;4;render_sun_cascade_2;0.2355 +21;3;accum_direct_blend;0.0000 +22;2;DEFER_SELF_ILLUM;0.0000 +23;2;DEFER_LIGHT;0.5704 +24;3;SHADOWED_LIGHTS;0.5059 +25;3;SPOT_LIGHTS_ACCUM_VOLUMETRIC;0.0614 +26;3;POINT_LIGHTS_ACCUM;0.0031 +27;2;DEFER_COMBINE;35.328 +28;3;CLOUDS_RENDER;0.0266 +29;3;combine_1;18.801 +30;3;render_forward;0.1373 +31;4;dsgraph_render_graph;0.0195 +32;5;dsgraph_render_static;0.0154 +33;5;dsgraph_render_dynamic;0.0041 +34;4;dsgraph_render_sorted;0.1167 +35;3;phase_combine_volumetric;0.0676 +36;3;render_distort_objects;0.0010 +37;4;dsgraph_render_distort;0.0000 +38;3;combine_distort;0.0664 +39;3;phase_bloom;0.0211 +40;3;combine_tonemap;0.0892 +41;3;DLSS;0.3965 +42;3;CAS;0.0635 +43;3;phase_ss_ss;0.2393 +44;4;phase_SS_SS_OGSE;0.2377 +45;5;copy_pp_ssss;0.0261 +46;3;PHASE_BLUR;0.1553 +47;3;phase_ssfx_bloom;0.0881 +48;3;combine_2;0.2482 +49;4;copy_pp_combine2;0.0264 +50;3;RenderFlares;0.0007 +51;3;phase_pp;0.0192 diff --git a/Benchmarks/cordon_pingpong2.csv b/Benchmarks/cordon_pingpong2.csv new file mode 100644 index 0000000000..d798b9d709 --- /dev/null +++ b/Benchmarks/cordon_pingpong2.csv @@ -0,0 +1,51 @@ +index;stack;name;time_ms +0;0;Frame;59.238 +1;1;CRender_Render;58.491 +2;2;SKY_RENDER;0.0440 +3;2;phase_scene_prepare;0.0061 +4;2;DEFER_PART0_SPLIT;0.4127 +5;3;dsgraph_render_static;0.3779 +6;3;dsgraph_render_dynamic;0.0348 +7;2;DEFER_TEST_LIGHT_VIS;0.0000 +8;2;DEFER_PART1_SPLIT;0.1782 +9;3;dsgraph_render_lods;0.0000 +10;3;DetailManager_Render;0.0778 +11;3;dsgraph_render_hud;0.1004 +12;2;copy_zbuffer;0.0222 +13;2;DEFER_WALLMARKS;0.0038 +14;3;dsgraph_render_wmarks;0.0031 +15;2;DEFER_RAIN;0.0000 +16;2;DEFER_SUN;13.251 +17;3;render_sun_flush;13.251 +18;4;render_sun_cascade_0;0.6154 +19;4;render_sun_cascade_1;0.4731 +20;4;render_sun_cascade_2;0.2355 +21;3;accum_direct_blend;0.0000 +22;2;DEFER_SELF_ILLUM;0.0000 +23;2;DEFER_LIGHT;0.5796 +24;3;SHADOWED_LIGHTS;0.5151 +25;3;SPOT_LIGHTS_ACCUM_VOLUMETRIC;0.0614 +26;3;POINT_LIGHTS_ACCUM;0.0031 +27;2;DEFER_COMBINE;32.748 +28;3;CLOUDS_RENDER;0.0266 +29;3;combine_1;17.848 +30;3;render_forward;0.0226 +31;4;dsgraph_render_graph;0.0205 +32;5;dsgraph_render_static;0.0164 +33;5;dsgraph_render_dynamic;0.0041 +34;4;dsgraph_render_sorted;0.0010 +35;3;phase_combine_volumetric;0.0727 +36;3;render_distort_objects;0.0020 +37;4;dsgraph_render_distort;0.0010 +38;3;combine_distort;0.0722 +39;3;phase_bloom;0.0204 +40;3;combine_tonemap;0.0889 +41;3;DLSS;0.3958 +42;3;CAS;0.0625 +43;3;phase_ss_ss;0.2130 +44;4;phase_SS_SS_OGSE;0.2120 +45;3;PHASE_BLUR;0.1567 +46;3;phase_ssfx_bloom;0.0881 +47;3;combine_2;0.2202 +48;3;RenderFlares;0.0000 +49;3;phase_pp;0.0174 diff --git a/Benchmarks/validate_ao_shaders.ps1 b/Benchmarks/validate_ao_shaders.ps1 new file mode 100644 index 0000000000..04c1779927 --- /dev/null +++ b/Benchmarks/validate_ao_shaders.ps1 @@ -0,0 +1,31 @@ +param( + [string]$Fxc = 'C:/Program Files (x86)/Windows Kits/10/bin/10.0.26100.0/x64/fxc.exe' +) + +$ErrorActionPreference = 'Stop' +$repoPath = Split-Path -Parent $PSScriptRoot +$shaderPath = Join-Path $repoPath 'Game/Resources_SoC_1.0006/gamedata/shaders/r3' +$outputPath = Join-Path $repoPath 'ogsr_engine/_TEMP/ao_validation' +New-Item -ItemType Directory -Force -Path $outputPath | Out-Null + +$compiledCount = 0 +foreach ($mode in @('ssdo', 'gtao')) { + foreach ($quality in 0..3) { + foreach ($shader in @('ogsr_ao', 'ogsr_ao_half', 'combine_1', 'combine_1_ao')) { + $shaderArgs = @('/nologo', '/T', 'ps_5_0', '/E', 'main', '/O1', '/Zpr') + if ($quality -gt 0) { $shaderArgs += @('/D', "SSAO_QUALITY=$quality") } + if ($mode -eq 'gtao') { $shaderArgs += @('/D', 'USE_GTAO=1') } + $shaderArgs += @('/Fo', (Join-Path $outputPath "$shader-$mode-$quality.cso"), (Join-Path $shaderPath "$shader.ps")) + $compilerOutput = & $Fxc @shaderArgs 2>&1 + if ($LASTEXITCODE -ne 0) { + throw "Failed: $shader, $mode, quality $quality`n$compilerOutput" + } + $compiledCount++ + } + } +} + +$compilerOutput = & $Fxc /nologo /T ps_5_0 /E main /O1 /Zpr /Fo (Join-Path $outputPath 'ogsr_ao_resolve.cso') (Join-Path $shaderPath 'ogsr_ao_resolve.ps') 2>&1 +if ($LASTEXITCODE -ne 0) { throw "Failed: AO resolve`n$compilerOutput" } +$compiledCount++ +Write-Output "PASS: $compiledCount shader variants (SSDO/GTAO, off/low/medium/high, legacy/separate/half and resolve)." diff --git a/Benchmarks/validate_xegtao_shaders.ps1 b/Benchmarks/validate_xegtao_shaders.ps1 new file mode 100644 index 0000000000..942bd3453b --- /dev/null +++ b/Benchmarks/validate_xegtao_shaders.ps1 @@ -0,0 +1,40 @@ +param( + [string]$Fxc = 'C:/Program Files (x86)/Windows Kits/10/bin/10.0.26100.0/x64/fxc.exe' +) + +# Optional, user-run compilation check. This does not build the engine or run GPU tests. +$ErrorActionPreference = 'Stop' +$repoPath = Split-Path -Parent $PSScriptRoot +$shaderPath = Join-Path $repoPath 'Game/Resources_SoC_1.0006/gamedata/shaders/r3' + +# FXC accepts slash-separated paths that OGSR's VFS cannot resolve. Check the +# runtime convention too, including nested vendor headers, before compiling. +$includeFiles = @(Get-ChildItem -LiteralPath $shaderPath -File -Filter 'ogsr_xegtao*') + + @(Get-ChildItem -LiteralPath (Join-Path $shaderPath 'xegtao') -File | Where-Object Extension -In @('.h', '.hlsli')) +foreach ($includeFile in $includeFiles) { + $includeText = Get-Content -LiteralPath $includeFile.FullName -Raw + foreach ($includeMatch in [regex]::Matches($includeText, '#include\s+"([^"]+)"')) { + if ($includeMatch.Groups[1].Value.Contains('/')) { + throw "OGSR requires backslashes in shader includes: $($includeFile.Name): $($includeMatch.Groups[1].Value)" + } + } +} + +$outputPath = Join-Path $repoPath 'ogsr_engine/_TEMP/ao_validation' +New-Item -ItemType Directory -Force -Path $outputPath | Out-Null + +$compiledCount = 0 +foreach ($quality in 1..3) { + foreach ($shader in @('ogsr_xegtao_prefilter.cs', 'ogsr_xegtao_main.cs', 'ogsr_xegtao_denoise.cs', 'ogsr_xegtao_export.ps', 'combine_1_ao.ps')) { + $target = if ($shader.EndsWith('.cs')) { 'cs_5_0' } else { 'ps_5_0' } + $shaderArgs = @('/nologo', '/T', $target, '/E', 'main', '/O1', '/Zpr', '/I', $shaderPath, + '/D', "SSAO_QUALITY=$quality", '/Fo', (Join-Path $outputPath "$shader-xegtao-$quality.cso"), + (Join-Path $shaderPath $shader)) + $compilerOutput = & $Fxc @shaderArgs 2>&1 + if ($LASTEXITCODE -ne 0) { + throw "Failed: $shader, quality $quality`n$compilerOutput" + } + $compiledCount++ + } +} +Write-Output "PASS: $compiledCount XeGTAO shader variants (three quality levels, compute passes, export and composition)." diff --git a/Build_ReleaseTracy.cmd b/Build_ReleaseTracy.cmd new file mode 100644 index 0000000000..719e068528 --- /dev/null +++ b/Build_ReleaseTracy.cmd @@ -0,0 +1,35 @@ +@echo off +setlocal + +cd /d "%~dp0" + +set "CONFIGURATION_GA=ReleaseTracyProfiler" + +set "VSWHERE=%ProgramFiles(x86)%\Microsoft Visual Studio\Installer\vswhere.exe" +if not exist "%VSWHERE%" ( + echo vswhere.exe not found. Install VS 2022 with the Desktop C++ workload. + exit /b 1 +) + +for /f "usebackq delims=" %%i in (`"%VSWHERE%" -latest -requires Microsoft.Component.MSBuild -find MSBuild\**\Bin\MSBuild.exe`) do set "MSBUILD=%%i" +if not defined MSBUILD ( + echo MSBuild.exe not found. + exit /b 1 +) + +echo Building Engine.sln Release^|x64 with CONFIGURATION_GA=%CONFIGURATION_GA% +echo Using "%MSBUILD%" +echo. + +"%MSBUILD%" Engine.sln /t:Rebuild /p:Configuration=Release /p:Platform=x64 /m +set "ERR=%ERRORLEVEL%" + +echo. +if %ERR% neq 0 ( + echo Build failed. Errorlevel %ERR%. + exit /b %ERR% +) + +echo Build succeeded. Output: "%cd%\bin_x64" +echo CONFIGURATION_GA was only set for this script. +exit /b 0 diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine.s b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine.s index 24b55cecb5..d78ee385ce 100644 --- a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine.s +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine.s @@ -1,5 +1,5 @@ -function element_0(shader, t_base, t_second, t_detail) - shader:begin("stub_screen_space", "combine_1") +local function setup_combine(shader, pixel_shader) + shader:begin("stub_screen_space", pixel_shader) :fog(false) :zb(false, false) :blend(true, blend.invsrcalpha,blend.srcalpha) @@ -8,6 +8,7 @@ function element_0(shader, t_base, t_second, t_detail) shader:dx10texture("s_position", "$user$position") shader:dx10texture("s_diffuse", "$user$albedo") shader:dx10texture("s_accumulator", "$user$accum") + shader:dx10texture("s_ao", "$user$ao") shader:dx10texture("env_s0", "$user$env_s0") shader:dx10texture("env_s1", "$user$env_s1") shader:dx10texture("sky_s0", "$user$sky0") @@ -21,6 +22,14 @@ function element_0(shader, t_base, t_second, t_detail) shader:dx10sampler("smp_rtlinear") end +function element_0(shader, t_base, t_second, t_detail) + setup_combine(shader, "combine_1") +end + +function element_4(shader, t_base, t_second, t_detail) + setup_combine(shader, "combine_1_ao") +end + function element_1(shader, t_base, t_second, t_detail) shader:begin("stub_screen_space", "combine_distort") :fog(false) diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1.ps index d01646678f..575f81cfb4 100644 Binary files a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1.ps and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1_ao.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1_ao.ps new file mode 100644 index 0000000000..a9ab76a8cc Binary files /dev/null and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/combine_1_ao.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/contrast_adaptive_sharpening.s b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/contrast_adaptive_sharpening.s index 3c75e056d2..35fde1a199 100644 --- a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/contrast_adaptive_sharpening.s +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/contrast_adaptive_sharpening.s @@ -5,3 +5,12 @@ function element_0(shader, t_base, t_second, t_detail) shader:dx10texture("t_current", "$user$postprocess0") shader:dx10sampler("SamplerLinearClamp") end + +-- Samples $user$generic_combine (DLSS/FSR output) instead of postprocess0. +function element_1(shader, t_base, t_second, t_detail) + shader:begin("stub_screen_space", "postprocess_cas") + :fog(false) + :zb(false, false) + shader:dx10texture("t_current", "$user$generic_combine") + shader:dx10sampler("SamplerLinearClamp") +end diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/hmodel.h b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/hmodel.h index 4bcb817830..8bc88e4163 100644 --- a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/hmodel.h +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/hmodel.h @@ -16,7 +16,7 @@ TextureCube env_s1; uniform float4 env_color; // color.w = lerp factor -void hmodel(out float3 hdiffuse, out float3 hspecular, float m, float h, float4 alb_gloss, float3 Pnt, float3 normal) +void hmodel(out float3 hdiffuse, out float3 hspecular, float m, float h, float4 alb_gloss, float3 Pnt, float3 normal, float3 ambientNormal) { // [ SSS Test ]. Overwrite terrain material bool m_terrain = abs(m - 0.95) <= 0.04f; @@ -74,21 +74,36 @@ void hmodel(out float3 hdiffuse, out float3 hspecular, float m, float h, float4 vreflectRemap = getSpecularDominantDir(nwRemap, vreflectRemap, rough); // Valve style ambient cube to prevent seams + // Only diffuse environment sampling uses the bent normal. The reflection + // direction, specular dominant direction and BRDF keep the surface normal. +#ifdef USE_XEGTAO_BENT_NORMALS + float3 ambientWorld = mul(m_inv_V, normalize(ambientNormal)); + float3 ambientRemap = ambientWorld; + float3 ambientAbs = abs(ambientRemap); + ambientRemap /= max(ambientAbs.x, max(ambientAbs.y, ambientAbs.z)); + if (ambientRemap.y < 0.999) + ambientRemap.y = ambientRemap.y * 2 - 1; + ambientRemap = normalize(ambientRemap); +#else + float3 ambientWorld = nw; + float3 ambientRemap = nwRemap; +#endif + const float Epsilon = 0.001; - float3 nSquared = nw * nw; + float3 nSquared = ambientWorld * ambientWorld; float3 e0d = 0; - e0d += nSquared.x * (env_s0.SampleLevel(smp_base, float3(nwRemap.x, Epsilon, Epsilon), CUBE_MIPS).rgb); - e0d += nSquared.y * (env_s0.SampleLevel(smp_base, float3(Epsilon, nwRemap.y, Epsilon), CUBE_MIPS).rgb); - e0d += nSquared.z * (env_s0.SampleLevel(smp_base, float3(Epsilon, Epsilon, nwRemap.z), CUBE_MIPS).rgb); + e0d += nSquared.x * (env_s0.SampleLevel(smp_base, float3(ambientRemap.x, Epsilon, Epsilon), CUBE_MIPS).rgb); + e0d += nSquared.y * (env_s0.SampleLevel(smp_base, float3(Epsilon, ambientRemap.y, Epsilon), CUBE_MIPS).rgb); + e0d += nSquared.z * (env_s0.SampleLevel(smp_base, float3(Epsilon, Epsilon, ambientRemap.z), CUBE_MIPS).rgb); // e0d = LinearTosRGB(e0d); // e0d = env_s0.SampleLevel(smp_base, nwRemap, CUBE_MIPS); float3 e1d = 0; - e1d += nSquared.x * (env_s1.SampleLevel(smp_base, float3(nwRemap.x, Epsilon, Epsilon), CUBE_MIPS).rgb); - e1d += nSquared.y * (env_s1.SampleLevel(smp_base, float3(Epsilon, nwRemap.y, Epsilon), CUBE_MIPS).rgb); - e1d += nSquared.z * (env_s1.SampleLevel(smp_base, float3(Epsilon, Epsilon, nwRemap.z), CUBE_MIPS).rgb); + e1d += nSquared.x * (env_s1.SampleLevel(smp_base, float3(ambientRemap.x, Epsilon, Epsilon), CUBE_MIPS).rgb); + e1d += nSquared.y * (env_s1.SampleLevel(smp_base, float3(Epsilon, ambientRemap.y, Epsilon), CUBE_MIPS).rgb); + e1d += nSquared.z * (env_s1.SampleLevel(smp_base, float3(Epsilon, Epsilon, ambientRemap.z), CUBE_MIPS).rgb); // e1d = LinearTosRGB(e1d); // e1d = env_s1.SampleLevel(smp_base, nwRemap, CUBE_MIPS); @@ -131,4 +146,10 @@ void hmodel(out float3 hdiffuse, out float3 hspecular, float m, float h, float4 hdiffuse = Amb_BRDF(rough, albedo, specular, env_d, env_s * !m_flora, -v2Pnt, nw).rgb; hspecular = 0; // do not use hspec at all } -#endif \ No newline at end of file + +// Existing callers (including forward materials) retain their original lighting. +void hmodel(out float3 hdiffuse, out float3 hspecular, float m, float h, float4 alb_gloss, float3 Pnt, float3 normal) +{ + hmodel(hdiffuse, hspecular, m, h, alb_gloss, Pnt, normal, normal); +} +#endif diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.ps new file mode 100644 index 0000000000..4abaaddf76 Binary files /dev/null and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.s b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.s new file mode 100644 index 0000000000..82d14bcff6 --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao.s @@ -0,0 +1,20 @@ +local function setup_ao(shader, pixel_shader) + shader:begin("stub_screen_space", pixel_shader) + :fog(false) + :zb(false, false) + shader:dx10texture("s_position", "$user$position") + shader:dx10sampler("smp_nofilter") +end + +function element_0(shader, t_base, t_second, t_detail) + setup_ao(shader, "ogsr_ao") +end + +function element_1(shader, t_base, t_second, t_detail) + setup_ao(shader, "ogsr_ao_half") +end + +function element_2(shader, t_base, t_second, t_detail) + setup_ao(shader, "ogsr_ao_resolve") + shader:dx10texture("s_ao_half", "$user$ao_half") +end diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_half.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_half.ps new file mode 100644 index 0000000000..ab050156dc Binary files /dev/null and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_half.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_resolve.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_resolve.ps new file mode 100644 index 0000000000..dbb2fd6d0a Binary files /dev/null and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_ao_resolve.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao.s b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao.s new file mode 100644 index 0000000000..2cf91dc5f1 --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao.s @@ -0,0 +1,7 @@ +function element_0(shader, t_base, t_second, t_detail) + shader:begin("stub_screen_space", "ogsr_xegtao_export") + :fog(false) + :zb(false, false) + shader:dx10texture("s_position", "$user$position") + shader:dx10texture("s_xegtao", "$user$xegtao") +end diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_common.h b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_common.h new file mode 100644 index 0000000000..7810ba312e --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_common.h @@ -0,0 +1,48 @@ +#ifndef OGSR_XEGTAO_COMMON_H +#define OGSR_XEGTAO_COMMON_H + +// SM5 / FXC path: FP32 arithmetic, R16F depth, R8 AO or packed R32 bent AO. +#define XE_GTAO_USE_HALF_FLOAT_PRECISION 0 +#ifdef USE_XEGTAO_BENT_NORMALS +#define XE_GTAO_COMPUTE_BENT_NORMALS +#endif +#define VA_SATURATE saturate +#include "xegtao\XeGTAO.hlsli" + +// Keep reflection flat: OGSR's constant parser does not support cbuffer structs. +cbuffer XeConstants : register(b0) +{ + float4 xe_viewport; // padded size, inverse padded size + float4 xe_source; // actual render size, noise frame, radius + float4 xe_ndc; // NDC-to-view multiply, add (including TAA jitter) +}; + +GTAOConstants XeConstantsForFrame() +{ + GTAOConstants c = (GTAOConstants)0; + c.ViewportSize = int2(xe_viewport.xy); + c.ViewportPixelSize = xe_viewport.zw; + c.NDCToViewMul = xe_ndc.xy; + c.NDCToViewAdd = xe_ndc.zw; + c.NDCToViewMul_x_PixelSize = xe_ndc.xy * xe_viewport.zw; + c.EffectRadius = xe_source.w; + c.EffectFalloffRange = XE_GTAO_DEFAULT_FALLOFF_RANGE; + c.RadiusMultiplier = XE_GTAO_DEFAULT_RADIUS_MULTIPLIER; + c.FinalValuePower = XE_GTAO_DEFAULT_FINAL_VALUE_POWER; + c.DenoiseBlurBeta = 1.2; + c.SampleDistributionPower = XE_GTAO_DEFAULT_SAMPLE_DISTRIBUTION_POWER; + c.ThinOccluderCompensation = XE_GTAO_DEFAULT_THIN_OCCLUDER_COMPENSATION; + c.DepthMIPSamplingOffset = XE_GTAO_DEFAULT_DEPTH_MIP_SAMPLING_OFFSET; + c.NoiseIndex = int(xe_source.z); + return c; +} + +float3 XeDecodeNormal(float2 packed) +{ + float2 f = packed * 2.0 - 1.0; + float3 n = float3(f, 1.0 - abs(f.x) - abs(f.y)); + float t = saturate(-n.z); + n.xy += float2(n.x >= 0 ? -t : t, n.y >= 0 ? -t : t); + return normalize(n); +} +#endif diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_denoise.cs b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_denoise.cs new file mode 100644 index 0000000000..d6f5642e06 --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_denoise.cs @@ -0,0 +1,13 @@ +#include "ogsr_xegtao_common.h" + +Texture2D xe_ao : register(t0); +// Match XeGTAO_Denoise exactly; FXC distinguishes lpfloat in texture templates. +Texture2D xe_edges : register(t1); +SamplerState xe_point : register(s0); +RWTexture2D xe_output : register(u0); + +[numthreads(8, 8, 1)] +void main(uint2 pixel : SV_DispatchThreadID) +{ + XeGTAO_Denoise(pixel * uint2(2, 1), XeConstantsForFrame(), xe_ao, xe_edges, xe_point, xe_output, true); +} diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_export.ps b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_export.ps new file mode 100644 index 0000000000..b1704cc5ff Binary files /dev/null and b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_export.ps differ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_main.cs b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_main.cs new file mode 100644 index 0000000000..13fe642821 --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_main.cs @@ -0,0 +1,32 @@ +#include "ogsr_xegtao_common.h" + +Texture2D xe_position : register(t0); +// Keep the template argument identical to XeGTAO_MainPass. FXC does not +// consider Texture2D interchangeable with Texture2D. +Texture2D xe_depth : register(t1); +SamplerState xe_point : register(s0); +RWTexture2D xe_ao : register(u0); +RWTexture2D xe_edges : register(u1); + +[numthreads(8, 8, 1)] +void main(uint2 pixel : SV_DispatchThreadID) +{ + float4 packed = xe_position.Load(int3(min(pixel, uint2(xe_source.xy) - 1), 0)); + if (packed.z <= 0.001) + { + XeGTAO_OutputWorkingTerm(pixel, 1.0, float3(0, 0, -1), xe_ao); + xe_edges[pixel] = 0; + return; + } + uint index = HilbertIndex(pixel.x % 64, pixel.y % 64) + 288 * uint(xe_source.z); + float2 noise = frac(0.5 + index * float2(0.75487766624669276, 0.56984029099805327)); +#if SSAO_QUALITY >= 3 + const float slices = 3, steps = 3; +#elif SSAO_QUALITY == 2 + const float slices = 2, steps = 2; +#else + const float slices = 1, steps = 2; +#endif + XeGTAO_MainPass(pixel, slices, steps, noise, XeDecodeNormal(packed.xy), + XeConstantsForFrame(), xe_depth, xe_point, xe_ao, xe_edges); +} diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_prefilter.cs b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_prefilter.cs new file mode 100644 index 0000000000..d808b29fcd --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/ogsr_xegtao_prefilter.cs @@ -0,0 +1,84 @@ +// Adapted from Intel XeGTAO (MIT); see xegtao/LICENSE. +#include "ogsr_xegtao_common.h" +Texture2D xe_position : register(t0); +RWTexture2D outDepth0 : register(u0); +RWTexture2D outDepth1 : register(u1); +RWTexture2D outDepth2 : register(u2); +RWTexture2D outDepth3 : register(u3); +RWTexture2D outDepth4 : register(u4); + +float XeLoadDepth(uint2 pixel) +{ + float depth = xe_position.Load(int3(min(pixel, uint2(xe_source.xy) - 1), 0)).z; + return depth > 0.001 ? min(depth, 65504.0) : 65504.0; +} + +[numthreads(8, 8, 1)] +void main(uint2 dispatchThreadID : SV_DispatchThreadID, uint2 groupThreadID : SV_GroupThreadID) +{ + const GTAOConstants consts = XeConstantsForFrame(); + // MIP 0 + const uint2 baseCoord = dispatchThreadID; + const uint2 pixCoord = baseCoord * 2; + lpfloat depth0 = XeLoadDepth(pixCoord + uint2(0, 0)); + lpfloat depth1 = XeLoadDepth(pixCoord + uint2(1, 0)); + lpfloat depth2 = XeLoadDepth(pixCoord + uint2(0, 1)); + lpfloat depth3 = XeLoadDepth(pixCoord + uint2(1, 1)); + outDepth0[ pixCoord + uint2(0, 0) ] = (lpfloat)depth0; + outDepth0[ pixCoord + uint2(1, 0) ] = (lpfloat)depth1; + outDepth0[ pixCoord + uint2(0, 1) ] = (lpfloat)depth2; + outDepth0[ pixCoord + uint2(1, 1) ] = (lpfloat)depth3; + + // MIP 1 + lpfloat dm1 = XeGTAO_DepthMIPFilter( depth0, depth1, depth2, depth3, consts ); + outDepth1[ baseCoord ] = (lpfloat)dm1; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm1; + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 2 + [branch] + if( all( ( groupThreadID.xy % uint2(2, 2) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+1][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+1]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+1][groupThreadID.y+1]; + + lpfloat dm2 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth2[ baseCoord / 2 ] = (lpfloat)dm2; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm2; + } + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 3 + [branch] + if( all( ( groupThreadID.xy % uint2(4, 4) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+2][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+2]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+2][groupThreadID.y+2]; + + lpfloat dm3 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth3[ baseCoord / 4 ] = (lpfloat)dm3; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm3; + } + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 4 + [branch] + if( all( ( groupThreadID.xy % uint2(8, 8) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+4][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+4]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+4][groupThreadID.y+4]; + + lpfloat dm4 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth4[ baseCoord / 8 ] = (lpfloat)dm4; + //g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm4; + } +} diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/LICENSE b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/LICENSE new file mode 100644 index 0000000000..601c6ef247 --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/LICENSE @@ -0,0 +1,22 @@ +MIT License + +Copyright (C) 2016-2021, Intel Corporation + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. + diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/README.md b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/README.md new file mode 100644 index 0000000000..d645bad75a --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/README.md @@ -0,0 +1,34 @@ +# XeGTAO + +MIT-licensed Intel XeGTAO from https://github.com/GameTechDev/XeGTAO, +pinned to commit `a5b1686c7ea37788eeb3576b5be47f7c03db532c`. + +`XeGTAO.h` and `XeGTAO.hlsli` originate in `Source/Rendering/Shaders/`. +Local modifications to `XeGTAO.hlsli`: the root-qualified include +`xegtao\XeGTAO.h`, required by OGSR's shader include resolver, and an explicit +saturate before converting denoised visibility to R8_UINT (the final rescale +can overshoot 1). Scalar swizzles are expanded to explicit vectors because +OGSR's FXC rejects forms such as `2.xx`. The same expansion is applied in +the adapted prefilter body. The wrapper uses XeGTAO's `lpfloat` texture type +verbatim because FXC treats it as distinct from `float` in texture templates. +`XeGTAO.h` is unchanged. See `LICENSE`. + +Use backslashes in shader include paths: OGSR's virtual filesystem does not +normalize forward slashes during lookup, even when Windows/FXC accepts them. + +OGSR adapters live in the parent directory as `ogsr_xegtao*`. The prefilter +adapts the upstream 16x16 routine to read linear view-space depth from OGSR's +G-buffer instead of converting hardware depth. Compute textures are padded +to multiples of 16 and border-filled, preserving five complete mip levels +even at odd or very small resolutions. Reconstruction accounts for padding +and the engine's projection jitter. Normals come from the existing G-buffer. + +`r_xegtao_bent_normals on` (default off; requires `vid_restart`) enables the +upstream `XE_GTAO_COMPUTE_BENT_NORMALS` branch through OGSR's +`USE_XEGTAO_BENT_NORMALS` shader permutation. Working/filtered AO then uses +R32_UINT with XYZ direction in the low three bytes and visibility in the +high byte. Export writes RGBA8_UNORM (R visibility, GBA view-space direction +encoded from [-1,1] to [0,1]), preserving the upstream component precision. +Only diffuse environment sampling consumes the bent normal; specular directions +and the BRDF retain the surface normal. The visibility-only path keeps R8_UINT +working/filtered textures and R16F export. diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.h b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.h new file mode 100644 index 0000000000..1738e4b04c --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.h @@ -0,0 +1,263 @@ +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// +// Copyright (C) 2016-2021, Intel Corporation +// +// SPDX-License-Identifier: MIT +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// +// +// XeGTAO is based on GTAO/GTSO "Jimenez et al. / Practical Real-Time Strategies for Accurate Indirect Occlusion", +// https://www.activision.com/cdn/research/Practical_Real_Time_Strategies_for_Accurate_Indirect_Occlusion_NEW%20VERSION_COLOR.pdf +// +// Implementation: Filip Strugar (filip.strugar@intel.com), Steve Mccalla (\_/) +// Version: 1.02 (='.'=) +// Details: https://github.com/GameTechDev/XeGTAO (")_(") +// +// Version history: +// 1.00 (2021-08-09): Initial release +// 1.01 (2021-09-02): Fix for depth going to inf for 'far' depth buffer values that are out of fp16 range +// 1.02 (2021-09-03): More fast_acos use and made final horizon cos clamping optional (off by default): 3-4% perf boost +// 1.10 (2021-09-03): Added a couple of heuristics to combat over-darkening errors in certain scenarios +// 1.20 (2021-09-06): Optional normal from depth generation is now a standalone pass: no longer integrated into +// main XeGTAO pass to reduce complexity and allow reuse; also quality of generated normals improved +// 1.21 (2021-09-28): Replaced 'groupshared'-based denoiser with a slightly slower multi-pass one where a 2-pass new +// equals 1-pass old. However, 1-pass new is faster than the 1-pass old and enough when TAA enabled. +// 1.22 (2021-09-28): Added 'XeGTAO_' prefix to all local functions to avoid name clashes with various user codebases. +// 1.30 (2021-10-10): Added support for directional component (bent normals). +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// + +#ifndef __XE_GTAO_TYPES_H__ +#define __XE_GTAO_TYPES_H__ + +#ifdef __cplusplus + +#include + +namespace XeGTAO +{ + + // cpp<->hlsl mapping + struct Matrix4x4 { float m[16]; }; + struct Vector3 { float x,y,z; }; + struct Vector2 { float x,y; }; + struct Vector2i { int x,y; }; + typedef unsigned int uint; + +#else // #ifdef __cplusplus + + // cpp<->hlsl mapping + #define Matrix4x4 float4x4 + #define Vector3 float3 + #define Vector2 float2 + #define Vector2i int2 + +#endif + + // Global consts that need to be visible from both shader and cpp side + #define XE_GTAO_DEPTH_MIP_LEVELS 5 // this one is hard-coded to 5 for now + #define XE_GTAO_NUMTHREADS_X 8 // these can be changed + #define XE_GTAO_NUMTHREADS_Y 8 // these can be changed + + struct GTAOConstants + { + Vector2i ViewportSize; + Vector2 ViewportPixelSize; // .zw == 1.0 / ViewportSize.xy + + Vector2 DepthUnpackConsts; + Vector2 CameraTanHalfFOV; + + Vector2 NDCToViewMul; + Vector2 NDCToViewAdd; + + Vector2 NDCToViewMul_x_PixelSize; + float EffectRadius; // world (viewspace) maximum size of the shadow + float EffectFalloffRange; + + float RadiusMultiplier; + float Padding0; + float FinalValuePower; + float DenoiseBlurBeta; + + float SampleDistributionPower; + float ThinOccluderCompensation; + float DepthMIPSamplingOffset; + int NoiseIndex; // frameIndex % 64 if using TAA or 0 otherwise + }; + + // This is used only for the development (ray traced ground truth). + struct ReferenceRTAOConstants + { + float TotalRaysLength ; // similar to Radius from GTAO + float Albedo ; // the assumption on the average material albedo + int MaxBounces ; // how many rays to recurse before stopping + int AccumulatedFrames ; // how many frames have we accumulated so far (after resetting/clearing). If 0 - this is the first. + int AccumulateFrameMax ; // how many frames are we aiming to accumulate; stop when we hit! + int Padding0; + int Padding1; + int Padding2; +#ifdef __cplusplus + ReferenceRTAOConstants( ) { TotalRaysLength = 1.0f; Albedo = 0.0f; MaxBounces = 1; AccumulatedFrames = 0; AccumulateFrameMax = 0; } +#endif + }; + + #ifndef XE_GTAO_USE_DEFAULT_CONSTANTS + #define XE_GTAO_USE_DEFAULT_CONSTANTS 1 + #endif + + // some constants reduce performance if provided as dynamic values; if these constants are not required to be dynamic and they match default values, + // set XE_GTAO_USE_DEFAULT_CONSTANTS and the code will compile into a more efficient shader + #define XE_GTAO_DEFAULT_RADIUS_MULTIPLIER (1.457f ) // allows us to use different value as compared to ground truth radius to counter inherent screen space biases + #define XE_GTAO_DEFAULT_FALLOFF_RANGE (0.615f ) // distant samples contribute less + #define XE_GTAO_DEFAULT_SAMPLE_DISTRIBUTION_POWER (2.0f ) // small crevices more important than big surfaces + #define XE_GTAO_DEFAULT_THIN_OCCLUDER_COMPENSATION (0.0f ) // the new 'thickness heuristic' approach + #define XE_GTAO_DEFAULT_FINAL_VALUE_POWER (2.2f ) // modifies the final ambient occlusion value using power function - this allows some of the above heuristics to do different things + #define XE_GTAO_DEFAULT_DEPTH_MIP_SAMPLING_OFFSET (3.30f ) // main trade-off between performance (memory bandwidth) and quality (temporal stability is the first affected, thin objects next) + + #define XE_GTAO_OCCLUSION_TERM_SCALE (1.5f) // for packing in UNORM (because raw, pre-denoised occlusion term can overshoot 1 but will later average out to 1) + + // From https://www.shadertoy.com/view/3tB3z3 - except we're using R2 here + #define XE_HILBERT_LEVEL 6U + #define XE_HILBERT_WIDTH ( (1U << XE_HILBERT_LEVEL) ) + #define XE_HILBERT_AREA ( XE_HILBERT_WIDTH * XE_HILBERT_WIDTH ) + inline uint HilbertIndex( uint posX, uint posY ) + { + uint index = 0U; + for( uint curLevel = XE_HILBERT_WIDTH/2U; curLevel > 0U; curLevel /= 2U ) + { + uint regionX = ( posX & curLevel ) > 0U; + uint regionY = ( posY & curLevel ) > 0U; + index += curLevel * curLevel * ( (3U * regionX) ^ regionY); + if( regionY == 0U ) + { + if( regionX == 1U ) + { + posX = uint( (XE_HILBERT_WIDTH - 1U) ) - posX; + posY = uint( (XE_HILBERT_WIDTH - 1U) ) - posY; + } + + uint temp = posX; + posX = posY; + posY = temp; + } + } + return index; + } + +#ifdef __cplusplus + + struct GTAOSettings + { + int QualityLevel = 2; // 0: low; 1: medium; 2: high; 3: ultra + int DenoisePasses = 1; // 0: disabled; 1: sharp; 2: medium; 3: soft + float Radius = 0.5f; // [0.0, ~ ] World (view) space size of the occlusion sphere. + + // auto-tune-d settings + float RadiusMultiplier = XE_GTAO_DEFAULT_RADIUS_MULTIPLIER; + float FalloffRange = XE_GTAO_DEFAULT_FALLOFF_RANGE; + float SampleDistributionPower = XE_GTAO_DEFAULT_SAMPLE_DISTRIBUTION_POWER; + float ThinOccluderCompensation = XE_GTAO_DEFAULT_THIN_OCCLUDER_COMPENSATION; + float FinalValuePower = XE_GTAO_DEFAULT_FINAL_VALUE_POWER; + float DepthMIPSamplingOffset = XE_GTAO_DEFAULT_DEPTH_MIP_SAMPLING_OFFSET; + }; + + template inline T clamp( T const & v, T const & min, T const & max ) { assert( max >= min ); if( v < min ) return min; if( v > max ) return max; return v; } + + // If using TAA then set noiseIndex to frameIndex % 64 - otherwise use 0 + inline void GTAOUpdateConstants( XeGTAO::GTAOConstants& consts, int viewportWidth, int viewportHeight, const XeGTAO::GTAOSettings & settings, const float projMatrix[16], bool rowMajor, unsigned int frameCounter ) + { + consts.ViewportSize = { viewportWidth, viewportHeight }; + consts.ViewportPixelSize = { 1.0f / (float)viewportWidth, 1.0f / (float)viewportHeight }; + + float depthLinearizeMul = (rowMajor)?(-projMatrix[3 * 4 + 2]):(-projMatrix[3 + 2 * 4]); // float depthLinearizeMul = ( clipFar * clipNear ) / ( clipFar - clipNear ); + float depthLinearizeAdd = (rowMajor)?( projMatrix[2 * 4 + 2]):( projMatrix[2 + 2 * 4]); // float depthLinearizeAdd = clipFar / ( clipFar - clipNear ); + + // correct the handedness issue. need to make sure this below is correct, but I think it is. + if( depthLinearizeMul * depthLinearizeAdd < 0 ) + depthLinearizeAdd = -depthLinearizeAdd; + consts.DepthUnpackConsts = { depthLinearizeMul, depthLinearizeAdd }; + + float tanHalfFOVY = 1.0f / ((rowMajor)?(projMatrix[1 * 4 + 1]):(projMatrix[1 + 1 * 4])); // = tanf( drawContext.Camera.GetYFOV( ) * 0.5f ); + float tanHalfFOVX = 1.0F / ((rowMajor)?(projMatrix[0 * 4 + 0]):(projMatrix[0 + 0 * 4])); // = tanHalfFOVY * drawContext.Camera.GetAspect( ); + consts.CameraTanHalfFOV = { tanHalfFOVX, tanHalfFOVY }; + + consts.NDCToViewMul = { consts.CameraTanHalfFOV.x * 2.0f, consts.CameraTanHalfFOV.y * -2.0f }; + consts.NDCToViewAdd = { consts.CameraTanHalfFOV.x * -1.0f, consts.CameraTanHalfFOV.y * 1.0f }; + + consts.NDCToViewMul_x_PixelSize = { consts.NDCToViewMul.x * consts.ViewportPixelSize.x, consts.NDCToViewMul.y * consts.ViewportPixelSize.y }; + + consts.EffectRadius = settings.Radius; + + consts.EffectFalloffRange = settings.FalloffRange; + consts.DenoiseBlurBeta = (settings.DenoisePasses==0)?(1e4f):(1.2f); // high value disables denoise - more elegant & correct way would be do set all edges to 0 + + consts.RadiusMultiplier = settings.RadiusMultiplier; + consts.SampleDistributionPower = settings.SampleDistributionPower; + consts.ThinOccluderCompensation = settings.ThinOccluderCompensation; + consts.FinalValuePower = settings.FinalValuePower; + consts.DepthMIPSamplingOffset = settings.DepthMIPSamplingOffset; + consts.NoiseIndex = (settings.DenoisePasses>0)?(frameCounter % 64):(0); + consts.Padding0 = 0; + } + +#ifdef IMGUI_API + inline bool GTAOImGuiSettings( XeGTAO::GTAOSettings & settings ) + { + bool hadChanges = false; + + ImGui::PushItemWidth( 120.0f ); + + ImGui::Text( "Performance/quality settings:" ); + + ImGui::Combo( "Quality Level", &settings.QualityLevel, "Low\0Medium\0High\0Ultra\00"); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Higher quality settings use more samples per pixel but are slower" ); + settings.QualityLevel = clamp( settings.QualityLevel , 0, 3 ); + + ImGui::Combo( "Denoising level", &settings.DenoisePasses, "Disabled\0Sharp\0Medium\0Soft\00"); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "The amount of edge-aware spatial denoise applied" ); + settings.DenoisePasses = clamp( settings.DenoisePasses , 0, 3 ); + + ImGui::Text( "Visual settings:" ); + + settings.Radius = clamp( settings.Radius, 0.0f, 100000.0f ); + + hadChanges |= ImGui::InputFloat( "Effect radius", &settings.Radius , 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "World (viewspace) effect radius\nExpected range: depends on the scene & requirements, anything from 0.01 to 1000+" ); + settings.Radius = clamp( settings.Radius , 0.0f, 10000.0f ); + + if( ImGui::CollapsingHeader( "Auto-tuned settings (heuristics)" ) ) + { + hadChanges |= ImGui::InputFloat( "Radius multiplier", &settings.RadiusMultiplier , 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Multiplies the 'Effect Radius' - used by the auto-tune to best match raytraced ground truth\nExpected range: [0.3, 3.0], defaults to %.3f", XE_GTAO_DEFAULT_RADIUS_MULTIPLIER ); + settings.RadiusMultiplier = clamp( settings.RadiusMultiplier , 0.3f, 3.0f ); + + hadChanges |= ImGui::InputFloat( "Falloff range", &settings.FalloffRange , 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Gently reduce sample impact as it gets out of 'Effect radius' bounds\nExpected range: [0.0, 1.0], defaults to %.3f", XE_GTAO_DEFAULT_FALLOFF_RANGE ); + settings.FalloffRange = clamp( settings.FalloffRange , 0.0f, 1.0f ); + + hadChanges |= ImGui::InputFloat( "Sample distribution power", &settings.SampleDistributionPower , 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Make samples on a slice equally distributed (1.0) or focus more towards the center (>1.0)\nExpected range: [1.0, 3.0], 2defaults to %.3f", XE_GTAO_DEFAULT_SAMPLE_DISTRIBUTION_POWER ); + settings.SampleDistributionPower = clamp( settings.SampleDistributionPower , 1.0f, 3.0f ); + + hadChanges |= ImGui::InputFloat( "Thin occluder compensation", &settings.ThinOccluderCompensation, 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Slightly reduce impact of samples further back to counter the bias from depth-based (incomplete) input scene geometry data\nExpected range: [0.0, 0.7], defaults to %.3f", XE_GTAO_DEFAULT_THIN_OCCLUDER_COMPENSATION ); + settings.ThinOccluderCompensation = clamp( settings.ThinOccluderCompensation , 0.0f, 0.7f ); + + hadChanges |= ImGui::InputFloat( "Final power", &settings.FinalValuePower, 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Applies power function to the final value: occlusion = pow( occlusion, finalPower )\nExpected range: [0.5, 5.0], defaults to %.3f", XE_GTAO_DEFAULT_FINAL_VALUE_POWER ); + settings.FinalValuePower = clamp( settings.FinalValuePower , 0.5f, 5.0f ); + + hadChanges |= ImGui::InputFloat( "Depth MIP sampling offset", &settings.DepthMIPSamplingOffset, 0.05f, 0.0f, "%.2f" ); + if( ImGui::IsItemHovered( ) ) ImGui::SetTooltip( "Mainly performance (texture memory bandwidth) setting but as a side-effect reduces overshadowing by thin objects and increases temporal instability\nExpected range: [2.0, 6.0], defaults to %.3f", XE_GTAO_DEFAULT_DEPTH_MIP_SAMPLING_OFFSET ); + settings.DepthMIPSamplingOffset = clamp( settings.DepthMIPSamplingOffset , 0.0f, 30.0f ); + } + + ImGui::PopItemWidth( ); + + return hadChanges; + } +#endif // IMGUI_API + +} // close the namespace + +#endif // #ifdef __cplusplus + + +#endif // __XE_GTAO_TYPES_H__ diff --git a/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.hlsli b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.hlsli new file mode 100644 index 0000000000..df21d2586a --- /dev/null +++ b/Game/Resources_SoC_1.0006/gamedata/shaders/r3/xegtao/XeGTAO.hlsli @@ -0,0 +1,855 @@ +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// +// Copyright (C) 2016-2021, Intel Corporation +// +// SPDX-License-Identifier: MIT +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// +// +// XeGTAO is based on GTAO/GTSO "Jimenez et al. / Practical Real-Time Strategies for Accurate Indirect Occlusion", +// https://www.activision.com/cdn/research/Practical_Real_Time_Strategies_for_Accurate_Indirect_Occlusion_NEW%20VERSION_COLOR.pdf +// +// Implementation: Filip Strugar (filip.strugar@intel.com), Steve Mccalla (\_/) +// Version: (see XeGTAO.h) (='.'=) +// Details: https://github.com/GameTechDev/XeGTAO (")_(") +// +// Version history: see XeGTAO.h +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ +#include "vaShared.hlsl" +#endif + +#if defined( XE_GTAO_SHOW_NORMALS ) || defined( XE_GTAO_SHOW_EDGES ) || defined( XE_GTAO_SHOW_BENT_NORMALS ) +RWTexture2D g_outputDbgImage : register( u2 ); +#endif + +#include "xegtao\XeGTAO.h" + +#define XE_GTAO_PI (3.1415926535897932384626433832795) +#define XE_GTAO_PI_HALF (1.5707963267948966192313216916398) + +#ifndef XE_GTAO_USE_HALF_FLOAT_PRECISION +#define XE_GTAO_USE_HALF_FLOAT_PRECISION 1 +#endif + +#if defined(XE_GTAO_FP32_DEPTHS) && XE_GTAO_USE_HALF_FLOAT_PRECISION +#error Using XE_GTAO_USE_HALF_FLOAT_PRECISION with 32bit depths is not supported yet unfortunately (it is possible to apply fp16 on parts not related to depth but this has not been done yet) +#endif + + +#if (XE_GTAO_USE_HALF_FLOAT_PRECISION != 0) +#if 1 // old fp16 approach ( float3 +float3 XeGTAO_R11G11B10_UNORM_to_FLOAT3( uint packedInput ) +{ + float3 unpackedOutput; + unpackedOutput.x = (float)( ( packedInput ) & 0x000007ff ) / 2047.0f; + unpackedOutput.y = (float)( ( packedInput >> 11 ) & 0x000007ff ) / 2047.0f; + unpackedOutput.z = (float)( ( packedInput >> 22 ) & 0x000003ff ) / 1023.0f; + return unpackedOutput; +} +// 'unpackedInput' is float3 and not float3 on purpose as half float lacks precision for below! +uint XeGTAO_FLOAT3_to_R11G11B10_UNORM( float3 unpackedInput ) +{ + uint packedOutput; + packedOutput =( ( uint( VA_SATURATE( unpackedInput.x ) * 2047 + 0.5f ) ) | + ( uint( VA_SATURATE( unpackedInput.y ) * 2047 + 0.5f ) << 11 ) | + ( uint( VA_SATURATE( unpackedInput.z ) * 1023 + 0.5f ) << 22 ) ); + return packedOutput; +} +// +lpfloat4 XeGTAO_R8G8B8A8_UNORM_to_FLOAT4( uint packedInput ) +{ + lpfloat4 unpackedOutput; + unpackedOutput.x = (lpfloat)( packedInput & 0x000000ff ) / (lpfloat)255; + unpackedOutput.y = (lpfloat)( ( ( packedInput >> 8 ) & 0x000000ff ) ) / (lpfloat)255; + unpackedOutput.z = (lpfloat)( ( ( packedInput >> 16 ) & 0x000000ff ) ) / (lpfloat)255; + unpackedOutput.w = (lpfloat)( packedInput >> 24 ) / (lpfloat)255; + return unpackedOutput; +} +// +uint XeGTAO_FLOAT4_to_R8G8B8A8_UNORM( lpfloat4 unpackedInput ) +{ + return (( uint( saturate( unpackedInput.x ) * (lpfloat)255 + (lpfloat)0.5 ) ) | + ( uint( saturate( unpackedInput.y ) * (lpfloat)255 + (lpfloat)0.5 ) << 8 ) | + ( uint( saturate( unpackedInput.z ) * (lpfloat)255 + (lpfloat)0.5 ) << 16 ) | + ( uint( saturate( unpackedInput.w ) * (lpfloat)255 + (lpfloat)0.5 ) << 24 ) ); +} + +/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////// + +// Inputs are screen XY and viewspace depth, output is viewspace position +float3 XeGTAO_ComputeViewspacePosition( const float2 screenPos, const float viewspaceDepth, const GTAOConstants consts ) +{ + float3 ret; + ret.xy = (consts.NDCToViewMul * screenPos.xy + consts.NDCToViewAdd) * viewspaceDepth; + ret.z = viewspaceDepth; + return ret; +} + +float XeGTAO_ScreenSpaceToViewSpaceDepth( const float screenDepth, const GTAOConstants consts ) +{ + float depthLinearizeMul = consts.DepthUnpackConsts.x; + float depthLinearizeAdd = consts.DepthUnpackConsts.y; + // Optimised version of "-cameraClipNear / (cameraClipFar - projDepth * (cameraClipFar - cameraClipNear)) * cameraClipFar" + return depthLinearizeMul / (depthLinearizeAdd - screenDepth); +} + +lpfloat4 XeGTAO_CalculateEdges( const lpfloat centerZ, const lpfloat leftZ, const lpfloat rightZ, const lpfloat topZ, const lpfloat bottomZ ) +{ + lpfloat4 edgesLRTB = lpfloat4( leftZ, rightZ, topZ, bottomZ ) - (lpfloat)centerZ; + + lpfloat slopeLR = (edgesLRTB.y - edgesLRTB.x) * 0.5; + lpfloat slopeTB = (edgesLRTB.w - edgesLRTB.z) * 0.5; + lpfloat4 edgesLRTBSlopeAdjusted = edgesLRTB + lpfloat4( slopeLR, -slopeLR, slopeTB, -slopeTB ); + edgesLRTB = min( abs( edgesLRTB ), abs( edgesLRTBSlopeAdjusted ) ); + return lpfloat4(saturate( ( 1.25 - edgesLRTB / (centerZ * 0.011) ) )); +} + +// packing/unpacking for edges; 2 bits per edge mean 4 gradient values (0, 0.33, 0.66, 1) for smoother transitions! +lpfloat XeGTAO_PackEdges( lpfloat4 edgesLRTB ) +{ + // integer version: + // edgesLRTB = saturate(edgesLRTB) * 2.9.xxxx + 0.5.xxxx; + // return (((uint)edgesLRTB.x) << 6) + (((uint)edgesLRTB.y) << 4) + (((uint)edgesLRTB.z) << 2) + (((uint)edgesLRTB.w)); + // + // optimized, should be same as above + edgesLRTB = round( saturate( edgesLRTB ) * 2.9 ); + return dot( edgesLRTB, lpfloat4( 64.0 / 255.0, 16.0 / 255.0, 4.0 / 255.0, 1.0 / 255.0 ) ) ; +} + +float3 XeGTAO_CalculateNormal( const float4 edgesLRTB, float3 pixCenterPos, float3 pixLPos, float3 pixRPos, float3 pixTPos, float3 pixBPos ) +{ + // Get this pixel's viewspace normal + float4 acceptedNormals = saturate( float4( edgesLRTB.x*edgesLRTB.z, edgesLRTB.z*edgesLRTB.y, edgesLRTB.y*edgesLRTB.w, edgesLRTB.w*edgesLRTB.x ) + 0.01 ); + + pixLPos = normalize(pixLPos - pixCenterPos); + pixRPos = normalize(pixRPos - pixCenterPos); + pixTPos = normalize(pixTPos - pixCenterPos); + pixBPos = normalize(pixBPos - pixCenterPos); + + float3 pixelNormal = acceptedNormals.x * cross( pixLPos, pixTPos ) + + + acceptedNormals.y * cross( pixTPos, pixRPos ) + + + acceptedNormals.z * cross( pixRPos, pixBPos ) + + + acceptedNormals.w * cross( pixBPos, pixLPos ); + pixelNormal = normalize( pixelNormal ); + + return pixelNormal; +} + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ +float4 DbgGetSliceColor(int slice, int sliceCount, bool mirror) +{ + float red = (float)slice / (float)sliceCount; float green = 0.01; float blue = 1.0 - (float)slice / (float)sliceCount; + return (mirror)?(float4(blue, green, red, 0.9)):(float4(red, green, blue, 0.9)); +} +#endif + +// http://h14s.p5r.org/2012/09/0x5f3759df.html, [Drobot2014a] Low Level Optimizations for GCN, https://blog.selfshadow.com/publications/s2016-shading-course/activision/s2016_pbs_activision_occlusion.pdf slide 63 +lpfloat XeGTAO_FastSqrt( float x ) +{ + return (lpfloat)(asfloat( 0x1fbd1df5 + ( asint( x ) >> 1 ) )); +} +// input [-1, 1] and output [0, PI], from https://seblagarde.wordpress.com/2014/12/01/inverse-trigonometric-functions-gpu-optimization-for-amd-gcn-architecture/ +lpfloat XeGTAO_FastACos( lpfloat inX ) +{ + const lpfloat PI = 3.141593; + const lpfloat HALF_PI = 1.570796; + lpfloat x = abs(inX); + lpfloat res = -0.156583 * x + HALF_PI; + res *= XeGTAO_FastSqrt(1.0 - x); + return (inX >= 0) ? res : PI - res; +} + +uint XeGTAO_EncodeVisibilityBentNormal( lpfloat visibility, lpfloat3 bentNormal ) +{ + return XeGTAO_FLOAT4_to_R8G8B8A8_UNORM( lpfloat4( bentNormal * 0.5 + 0.5, visibility ) ); +} + +void XeGTAO_DecodeVisibilityBentNormal( const uint packedValue, out lpfloat visibility, out lpfloat3 bentNormal ) +{ + lpfloat4 decoded = XeGTAO_R8G8B8A8_UNORM_to_FLOAT4( packedValue ); + bentNormal = decoded.xyz * float3(2.0, 2.0, 2.0) - float3(1.0, 1.0, 1.0); // could normalize - don't want to since it's done so many times, better to do it at the final step only + visibility = decoded.w; +} + +void XeGTAO_OutputWorkingTerm( const uint2 pixCoord, lpfloat visibility, lpfloat3 bentNormal, RWTexture2D outWorkingAOTerm ) +{ + visibility = saturate( visibility / lpfloat(XE_GTAO_OCCLUSION_TERM_SCALE) ); +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + outWorkingAOTerm[pixCoord] = XeGTAO_EncodeVisibilityBentNormal( visibility, bentNormal ); +#else + outWorkingAOTerm[pixCoord] = uint(visibility * 255.0 + 0.5); +#endif +} + +// "Efficiently building a matrix to rotate one vector to another" +// http://cs.brown.edu/research/pubs/pdfs/1999/Moller-1999-EBA.pdf / https://dl.acm.org/doi/10.1080/10867651.1999.10487509 +// (using https://github.com/assimp/assimp/blob/master/include/assimp/matrix3x3.inl#L275 as a code reference as it seems to be best) +lpfloat3x3 XeGTAO_RotFromToMatrix( lpfloat3 from, lpfloat3 to ) +{ + const lpfloat e = dot(from, to); + const lpfloat f = abs(e); //(e < 0)? -e:e; + + // WARNING: This has not been tested/worked through, especially not for 16bit floats; seems to work in our special use case (from is always {0, 0, -1}) but wouldn't use it in general + if( f > lpfloat( 1.0 - 0.0003 ) ) + return lpfloat3x3( 1, 0, 0, 0, 1, 0, 0, 0, 1 ); + + const lpfloat3 v = cross( from, to ); + /* ... use this hand optimized version (9 mults less) */ + const lpfloat h = (1.0)/(1.0 + e); /* optimization by Gottfried Chen */ + const lpfloat hvx = h * v.x; + const lpfloat hvz = h * v.z; + const lpfloat hvxy = hvx * v.y; + const lpfloat hvxz = hvx * v.z; + const lpfloat hvyz = hvz * v.y; + + lpfloat3x3 mtx; + mtx[0][0] = e + hvx * v.x; + mtx[0][1] = hvxy - v.z; + mtx[0][2] = hvxz + v.y; + + mtx[1][0] = hvxy + v.z; + mtx[1][1] = e + h * v.y * v.y; + mtx[1][2] = hvyz - v.x; + + mtx[2][0] = hvxz - v.y; + mtx[2][1] = hvyz + v.x; + mtx[2][2] = e + hvz * v.z; + + return mtx; +} + +void XeGTAO_MainPass( const uint2 pixCoord, lpfloat sliceCount, lpfloat stepsPerSlice, const lpfloat2 localNoise, lpfloat3 viewspaceNormal, const GTAOConstants consts, + Texture2D sourceViewspaceDepth, SamplerState depthSampler, RWTexture2D outWorkingAOTerm, RWTexture2D outWorkingEdges ) +{ + float2 normalizedScreenPos = (pixCoord + float2(0.5, 0.5)) * consts.ViewportPixelSize; + + lpfloat4 valuesUL = sourceViewspaceDepth.GatherRed( depthSampler, float2( pixCoord * consts.ViewportPixelSize ) ); + lpfloat4 valuesBR = sourceViewspaceDepth.GatherRed( depthSampler, float2( pixCoord * consts.ViewportPixelSize ), int2( 1, 1 ) ); + + // viewspace Z at the center + lpfloat viewspaceZ = valuesUL.y; //sourceViewspaceDepth.SampleLevel( depthSampler, normalizedScreenPos, 0 ).x; + + // viewspace Zs left top right bottom + const lpfloat pixLZ = valuesUL.x; + const lpfloat pixTZ = valuesUL.z; + const lpfloat pixRZ = valuesBR.z; + const lpfloat pixBZ = valuesBR.x; + + lpfloat4 edgesLRTB = XeGTAO_CalculateEdges( (lpfloat)viewspaceZ, (lpfloat)pixLZ, (lpfloat)pixRZ, (lpfloat)pixTZ, (lpfloat)pixBZ ); + outWorkingEdges[pixCoord] = XeGTAO_PackEdges(edgesLRTB); + + // Generating screen space normals in-place is faster than generating normals in a separate pass but requires + // use of 32bit depth buffer (16bit works but visibly degrades quality) which in turn slows everything down. So to + // reduce complexity and allow for screen space normal reuse by other effects, we've pulled it out into a separate + // pass. + // However, we leave this code in, in case anyone has a use-case where it fits better. +#ifdef XE_GTAO_GENERATE_NORMALS_INPLACE + float3 CENTER = XeGTAO_ComputeViewspacePosition( normalizedScreenPos, viewspaceZ, consts ); + float3 LEFT = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2(-1, 0) * consts.ViewportPixelSize, pixLZ, consts ); + float3 RIGHT = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 1, 0) * consts.ViewportPixelSize, pixRZ, consts ); + float3 TOP = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 0, -1) * consts.ViewportPixelSize, pixTZ, consts ); + float3 BOTTOM = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 0, 1) * consts.ViewportPixelSize, pixBZ, consts ); + viewspaceNormal = (lpfloat3)XeGTAO_CalculateNormal( edgesLRTB, CENTER, LEFT, RIGHT, TOP, BOTTOM ); +#endif + + // Move center pixel slightly towards camera to avoid imprecision artifacts due to depth buffer imprecision; offset depends on depth texture format used +#ifdef XE_GTAO_FP32_DEPTHS + viewspaceZ *= 0.99999; // this is good for FP32 depth buffer +#else + viewspaceZ *= 0.99920; // this is good for FP16 depth buffer +#endif + + const float3 pixCenterPos = XeGTAO_ComputeViewspacePosition( normalizedScreenPos, viewspaceZ, consts ); + const lpfloat3 viewVec = (lpfloat3)normalize(-pixCenterPos); + + // prevents normals that are facing away from the view vector - xeGTAO struggles with extreme cases, but in Vanilla it seems rare so it's disabled by default + // viewspaceNormal = normalize( viewspaceNormal + max( 0, -dot( viewspaceNormal, viewVec ) ) * viewVec ); + +#ifdef XE_GTAO_SHOW_NORMALS + g_outputDbgImage[pixCoord] = float4( DisplayNormalSRGB( viewspaceNormal.xyz ), 1 ); +#endif + +#ifdef XE_GTAO_SHOW_EDGES + g_outputDbgImage[pixCoord] = 1.0 - float4( edgesLRTB.x, edgesLRTB.y * 0.5 + edgesLRTB.w * 0.5, edgesLRTB.z, 1.0 ); +#endif + +#if XE_GTAO_USE_DEFAULT_CONSTANTS != 0 + const lpfloat effectRadius = (lpfloat)consts.EffectRadius * (lpfloat)XE_GTAO_DEFAULT_RADIUS_MULTIPLIER; + const lpfloat sampleDistributionPower = (lpfloat)XE_GTAO_DEFAULT_SAMPLE_DISTRIBUTION_POWER; + const lpfloat thinOccluderCompensation = (lpfloat)XE_GTAO_DEFAULT_THIN_OCCLUDER_COMPENSATION; + const lpfloat falloffRange = (lpfloat)XE_GTAO_DEFAULT_FALLOFF_RANGE * effectRadius; +#else + const lpfloat effectRadius = (lpfloat)consts.EffectRadius * (lpfloat)consts.RadiusMultiplier; + const lpfloat sampleDistributionPower = (lpfloat)consts.SampleDistributionPower; + const lpfloat thinOccluderCompensation = (lpfloat)consts.ThinOccluderCompensation; + const lpfloat falloffRange = (lpfloat)consts.EffectFalloffRange * effectRadius; +#endif + + const lpfloat falloffFrom = effectRadius * ((lpfloat)1-(lpfloat)consts.EffectFalloffRange); + + // fadeout precompute optimisation + const lpfloat falloffMul = (lpfloat)-1.0 / ( falloffRange ); + const lpfloat falloffAdd = falloffFrom / ( falloffRange ) + (lpfloat)1.0; + + lpfloat visibility = 0; +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + lpfloat3 bentNormal = 0; +#else + lpfloat3 bentNormal = viewspaceNormal; +#endif + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ + float3 dbgWorldPos = mul(g_globals.ViewInv, float4(pixCenterPos, 1)).xyz; +#endif + + // see "Algorithm 1" in https://www.activision.com/cdn/research/Practical_Real_Time_Strategies_for_Accurate_Indirect_Occlusion_NEW%20VERSION_COLOR.pdf + { + const lpfloat noiseSlice = (lpfloat)localNoise.x; + const lpfloat noiseSample = (lpfloat)localNoise.y; + + // quality settings / tweaks / hacks + const lpfloat pixelTooCloseThreshold = 1.3; // if the offset is under approx pixel size (pixelTooCloseThreshold), push it out to the minimum distance + + // approx viewspace pixel size at pixCoord; approximation of NDCToViewspace( normalizedScreenPos.xy + consts.ViewportPixelSize.xy, pixCenterPos.z ).xy - pixCenterPos.xy; + const float2 pixelDirRBViewspaceSizeAtCenterZ = viewspaceZ.xx * consts.NDCToViewMul_x_PixelSize; + + lpfloat screenspaceRadius = effectRadius / (lpfloat)pixelDirRBViewspaceSizeAtCenterZ.x; + + // fade out for small screen radii + visibility += saturate((10 - screenspaceRadius)/100)*0.5; + +#if 0 // sensible early-out for even more performance; disabled because not yet tested + [branch] + if( screenspaceRadius < pixelTooCloseThreshold ) + { + XeGTAO_OutputWorkingTerm( pixCoord, 1, viewspaceNormal, outWorkingAOTerm ); + return; + } +#endif + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ + [branch] if (IsUnderCursorRange(pixCoord, int2(1, 1))) + { + float3 dbgWorldNorm = mul((float3x3)g_globals.ViewInv, viewspaceNormal).xyz; + float3 dbgWorldViewVec = mul((float3x3)g_globals.ViewInv, viewVec).xyz; + //DebugDraw3DArrow(dbgWorldPos, dbgWorldPos + 0.5 * dbgWorldViewVec, 0.02, float4(0, 1, 0, 0.95)); + //DebugDraw2DCircle(pixCoord, screenspaceRadius, float4(1, 0, 0.2, 1)); + DebugDraw3DSphere(dbgWorldPos, effectRadius, float4(1, 0.2, 0, 0.1)); + //DebugDraw3DText(dbgWorldPos, float2(0, 0), float4(0.6, 0.3, 0.3, 1), float4( pixelDirRBViewspaceSizeAtCenterZ.xy, 0, screenspaceRadius) ); + } +#endif + + // this is the min distance to start sampling from to avoid sampling from the center pixel (no useful data obtained from sampling center pixel) + const lpfloat minS = (lpfloat)pixelTooCloseThreshold / screenspaceRadius; + + //[unroll] + for( lpfloat slice = 0; slice < sliceCount; slice++ ) + { + lpfloat sliceK = (slice+noiseSlice) / sliceCount; + // lines 5, 6 from the paper + lpfloat phi = sliceK * XE_GTAO_PI; + lpfloat cosPhi = cos(phi); + lpfloat sinPhi = sin(phi); + lpfloat2 omega = lpfloat2(cosPhi, -sinPhi); //lpfloat2 on omega causes issues with big radii + + // convert to screen units (pixels) for later use + omega *= screenspaceRadius; + + // line 8 from the paper + const lpfloat3 directionVec = lpfloat3(cosPhi, sinPhi, 0); + + // line 9 from the paper + const lpfloat3 orthoDirectionVec = directionVec - (dot(directionVec, viewVec) * viewVec); + + // line 10 from the paper + //axisVec is orthogonal to directionVec and viewVec, used to define projectedNormal + const lpfloat3 axisVec = normalize( cross(orthoDirectionVec, viewVec) ); + + // alternative line 9 from the paper + // float3 orthoDirectionVec = cross( viewVec, axisVec ); + + // line 11 from the paper + lpfloat3 projectedNormalVec = viewspaceNormal - axisVec * dot(viewspaceNormal, axisVec); + + // line 13 from the paper + lpfloat signNorm = (lpfloat)sign( dot( orthoDirectionVec, projectedNormalVec ) ); + + // line 14 from the paper + lpfloat projectedNormalVecLength = length(projectedNormalVec); + lpfloat cosNorm = (lpfloat)saturate(dot(projectedNormalVec, viewVec) / projectedNormalVecLength); + + // line 15 from the paper + lpfloat n = signNorm * XeGTAO_FastACos(cosNorm); + + // this is a lower weight target; not using -1 as in the original paper because it is under horizon, so a 'weight' has different meaning based on the normal + const lpfloat lowHorizonCos0 = cos(n+XE_GTAO_PI_HALF); + const lpfloat lowHorizonCos1 = cos(n-XE_GTAO_PI_HALF); + + // lines 17, 18 from the paper, manually unrolled the 'side' loop + lpfloat horizonCos0 = lowHorizonCos0; //-1; + lpfloat horizonCos1 = lowHorizonCos1; //-1; + + [unroll] + for( lpfloat step = 0; step < stepsPerSlice; step++ ) + { + // R1 sequence (http://extremelearning.com.au/unreasonable-effectiveness-of-quasirandom-sequences/) + const lpfloat stepBaseNoise = lpfloat(slice + step * stepsPerSlice) * 0.6180339887498948482; // <- this should unroll + lpfloat stepNoise = frac(noiseSample + stepBaseNoise); + + // approx line 20 from the paper, with added noise + lpfloat s = (step+stepNoise) / (stepsPerSlice); // + (lpfloat2)1e-6f); + + // additional distribution modifier + s = (lpfloat)pow( s, (lpfloat)sampleDistributionPower ); + + // avoid sampling center pixel + s += minS; + + // approx lines 21-22 from the paper, unrolled + lpfloat2 sampleOffset = s * omega; + + lpfloat sampleOffsetLength = length( sampleOffset ); + + // note: when sampling, using point_point_point or point_point_linear sampler works, but linear_linear_linear will cause unwanted interpolation between neighbouring depth values on the same MIP level! + const lpfloat mipLevel = (lpfloat)clamp( log2( sampleOffsetLength ) - consts.DepthMIPSamplingOffset, 0, XE_GTAO_DEPTH_MIP_LEVELS ); + + // Snap to pixel center (more correct direction math, avoids artifacts due to sampling pos not matching depth texel center - messes up slope - but adds other + // artifacts due to them being pushed off the slice). Also use full precision for high res cases. + sampleOffset = round(sampleOffset) * (lpfloat2)consts.ViewportPixelSize; + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ + int mipLevelU = (int)round(mipLevel); + float4 mipColor = saturate( float4( mipLevelU>=3, mipLevelU>=1 && mipLevelU<=3, mipLevelU<=1, 1.0 ) ); + if( all( sampleOffset == 0 ) ) + DebugDraw2DText( pixCoord, float4( 1, 0, 0, 1), pixelTooCloseThreshold ); + [branch] if (IsUnderCursorRange(pixCoord, int2(1, 1))) + { + //DebugDraw2DText( (normalizedScreenPos + sampleOffset) * consts.ViewportSize, mipColor, mipLevelU ); + //DebugDraw2DText( (normalizedScreenPos + sampleOffset) * consts.ViewportSize, mipColor, (uint)slice ); + //DebugDraw2DText( (normalizedScreenPos - sampleOffset) * consts.ViewportSize, mipColor, (uint)slice ); + //DebugDraw2DText( (normalizedScreenPos - sampleOffset) * consts.ViewportSize, saturate( float4( mipLevelU>=3, mipLevelU>=1 && mipLevelU<=3, mipLevelU<=1, 1.0 ) ), mipLevelU ); + } +#endif + + float2 sampleScreenPos0 = normalizedScreenPos + sampleOffset; + float SZ0 = sourceViewspaceDepth.SampleLevel( depthSampler, sampleScreenPos0, mipLevel ).x; + float3 samplePos0 = XeGTAO_ComputeViewspacePosition( sampleScreenPos0, SZ0, consts ); + + float2 sampleScreenPos1 = normalizedScreenPos - sampleOffset; + float SZ1 = sourceViewspaceDepth.SampleLevel( depthSampler, sampleScreenPos1, mipLevel ).x; + float3 samplePos1 = XeGTAO_ComputeViewspacePosition( sampleScreenPos1, SZ1, consts ); + + float3 sampleDelta0 = (samplePos0 - float3(pixCenterPos)); // using lpfloat for sampleDelta causes precision issues + float3 sampleDelta1 = (samplePos1 - float3(pixCenterPos)); // using lpfloat for sampleDelta causes precision issues + lpfloat sampleDist0 = (lpfloat)length( sampleDelta0 ); + lpfloat sampleDist1 = (lpfloat)length( sampleDelta1 ); + + // approx lines 23, 24 from the paper, unrolled + lpfloat3 sampleHorizonVec0 = (lpfloat3)(sampleDelta0 / sampleDist0); + lpfloat3 sampleHorizonVec1 = (lpfloat3)(sampleDelta1 / sampleDist1); + + // any sample out of radius should be discarded - also use fallof range for smooth transitions; this is a modified idea from "4.3 Implementation details, Bounding the sampling area" +#if XE_GTAO_USE_DEFAULT_CONSTANTS != 0 && XE_GTAO_DEFAULT_THIN_OBJECT_HEURISTIC == 0 + lpfloat weight0 = saturate( sampleDist0 * falloffMul + falloffAdd ); + lpfloat weight1 = saturate( sampleDist1 * falloffMul + falloffAdd ); +#else + // this is our own thickness heuristic that relies on sooner discarding samples behind the center + lpfloat falloffBase0 = length( lpfloat3(sampleDelta0.x, sampleDelta0.y, sampleDelta0.z * (1+thinOccluderCompensation) ) ); + lpfloat falloffBase1 = length( lpfloat3(sampleDelta1.x, sampleDelta1.y, sampleDelta1.z * (1+thinOccluderCompensation) ) ); + lpfloat weight0 = saturate( falloffBase0 * falloffMul + falloffAdd ); + lpfloat weight1 = saturate( falloffBase1 * falloffMul + falloffAdd ); +#endif + + // sample horizon cos + lpfloat shc0 = (lpfloat)dot(sampleHorizonVec0, viewVec); + lpfloat shc1 = (lpfloat)dot(sampleHorizonVec1, viewVec); + + // discard unwanted samples + shc0 = lerp( lowHorizonCos0, shc0, weight0 ); // this would be more correct but too expensive: cos(lerp( acos(lowHorizonCos0), acos(shc0), weight0 )); + shc1 = lerp( lowHorizonCos1, shc1, weight1 ); // this would be more correct but too expensive: cos(lerp( acos(lowHorizonCos1), acos(shc1), weight1 )); + + // thickness heuristic - see "4.3 Implementation details, Height-field assumption considerations" +#if 0 // (disabled, not used) this should match the paper + lpfloat newhorizonCos0 = max( horizonCos0, shc0 ); + lpfloat newhorizonCos1 = max( horizonCos1, shc1 ); + horizonCos0 = (horizonCos0 > shc0)?( lerp( newhorizonCos0, shc0, thinOccluderCompensation ) ):( newhorizonCos0 ); + horizonCos1 = (horizonCos1 > shc1)?( lerp( newhorizonCos1, shc1, thinOccluderCompensation ) ):( newhorizonCos1 ); +#elif 0 // (disabled, not used) this is slightly different from the paper but cheaper and provides very similar results + horizonCos0 = lerp( max( horizonCos0, shc0 ), shc0, thinOccluderCompensation ); + horizonCos1 = lerp( max( horizonCos1, shc1 ), shc1, thinOccluderCompensation ); +#else // this is a version where thicknessHeuristic is completely disabled + horizonCos0 = max( horizonCos0, shc0 ); + horizonCos1 = max( horizonCos1, shc1 ); +#endif + + +#ifdef XE_GTAO_SHOW_DEBUG_VIZ + [branch] if (IsUnderCursorRange(pixCoord, int2(1, 1))) + { + float3 WS_samplePos0 = mul(g_globals.ViewInv, float4(samplePos0, 1)).xyz; + float3 WS_samplePos1 = mul(g_globals.ViewInv, float4(samplePos1, 1)).xyz; + float3 WS_sampleHorizonVec0 = mul( (float3x3)g_globals.ViewInv, sampleHorizonVec0).xyz; + float3 WS_sampleHorizonVec1 = mul( (float3x3)g_globals.ViewInv, sampleHorizonVec1).xyz; + // DebugDraw3DSphere( WS_samplePos0, effectRadius * 0.02, DbgGetSliceColor(slice, sliceCount, false) ); + // DebugDraw3DSphere( WS_samplePos1, effectRadius * 0.02, DbgGetSliceColor(slice, sliceCount, true) ); + DebugDraw3DSphere( WS_samplePos0, effectRadius * 0.02, mipColor ); + DebugDraw3DSphere( WS_samplePos1, effectRadius * 0.02, mipColor ); + // DebugDraw3DArrow( WS_samplePos0, WS_samplePos0 - WS_sampleHorizonVec0, 0.002, float4(1, 0, 0, 1 ) ); + // DebugDraw3DArrow( WS_samplePos1, WS_samplePos1 - WS_sampleHorizonVec1, 0.002, float4(1, 0, 0, 1 ) ); + // DebugDraw3DText( WS_samplePos0, float2(0, 0), float4( 1, 0, 0, 1), weight0 ); + // DebugDraw3DText( WS_samplePos1, float2(0, 0), float4( 1, 0, 0, 1), weight1 ); + + // DebugDraw2DText( float2( 500, 94+(step+slice*3)*12 ), float4( 0, 1, 0, 1 ), float4( projectedNormalVecLength, 0, horizonCos0, horizonCos1 ) ); + } +#endif + } + +#if 1 // I can't figure out the slight overdarkening on high slopes, so I'm adding this fudge - in the training set, 0.05 is close (PSNR 21.34) to disabled (PSNR 21.45) + projectedNormalVecLength = lerp( projectedNormalVecLength, 1, 0.05 ); +#endif + + // line ~27, unrolled + lpfloat h0 = -XeGTAO_FastACos((lpfloat)horizonCos1); + lpfloat h1 = XeGTAO_FastACos((lpfloat)horizonCos0); +#if 0 // we can skip clamping for a tiny little bit more performance + h0 = n + clamp( h0-n, (lpfloat)-XE_GTAO_PI_HALF, (lpfloat)XE_GTAO_PI_HALF ); + h1 = n + clamp( h1-n, (lpfloat)-XE_GTAO_PI_HALF, (lpfloat)XE_GTAO_PI_HALF ); +#endif + lpfloat iarc0 = ((lpfloat)cosNorm + (lpfloat)2 * (lpfloat)h0 * (lpfloat)sin(n)-(lpfloat)cos((lpfloat)2 * (lpfloat)h0-n))/(lpfloat)4; + lpfloat iarc1 = ((lpfloat)cosNorm + (lpfloat)2 * (lpfloat)h1 * (lpfloat)sin(n)-(lpfloat)cos((lpfloat)2 * (lpfloat)h1-n))/(lpfloat)4; + lpfloat localVisibility = (lpfloat)projectedNormalVecLength * (lpfloat)(iarc0+iarc1); + visibility += localVisibility; + +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + // see "Algorithm 2 Extension that computes bent normals b." + lpfloat t0 = (6*sin(h0-n)-sin(3*h0-n)+6*sin(h1-n)-sin(3*h1-n)+16*sin(n)-3*(sin(h0+n)+sin(h1+n)))/12; + lpfloat t1 = (-cos(3 * h0-n)-cos(3 * h1-n) +8 * cos(n)-3 * (cos(h0+n) +cos(h1+n)))/12; + lpfloat3 localBentNormal = lpfloat3( directionVec.x * (lpfloat)t0, directionVec.y * (lpfloat)t0, -lpfloat(t1) ); + localBentNormal = (lpfloat3)mul( XeGTAO_RotFromToMatrix( lpfloat3(0,0,-1), viewVec ), localBentNormal ) * projectedNormalVecLength; + bentNormal += localBentNormal; +#endif + } + visibility /= (lpfloat)sliceCount; + visibility = pow( visibility, (lpfloat)consts.FinalValuePower ); + visibility = max( (lpfloat)0.03, visibility ); // disallow total occlusion (which wouldn't make any sense anyhow since pixel is visible but also helps with packing bent normals) + +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + bentNormal = normalize(bentNormal) ; +#endif + } + +#if defined(XE_GTAO_SHOW_DEBUG_VIZ) && defined(XE_GTAO_COMPUTE_BENT_NORMALS) + [branch] if (IsUnderCursorRange(pixCoord, int2(1, 1))) + { + float3 dbgWorldViewNorm = mul((float3x3)g_globals.ViewInv, viewspaceNormal).xyz; + float3 dbgWorldBentNorm = mul((float3x3)g_globals.ViewInv, bentNormal).xyz; + DebugDraw3DSphereCone( dbgWorldPos, dbgWorldViewNorm, 0.3, VA_PI*0.5 - acos(saturate(visibility)), float4( 0.2, 0.2, 0.2, 0.5 ) ); + DebugDraw3DSphereCone( dbgWorldPos, dbgWorldBentNorm, 0.3, VA_PI*0.5 - acos(saturate(visibility)), float4( 0.0, 1.0, 0.0, 0.7 ) ); + } +#endif + + XeGTAO_OutputWorkingTerm( pixCoord, visibility, bentNormal, outWorkingAOTerm ); +} + +// weighted average depth filter +lpfloat XeGTAO_DepthMIPFilter( lpfloat depth0, lpfloat depth1, lpfloat depth2, lpfloat depth3, const GTAOConstants consts ) +{ + lpfloat maxDepth = max( max( depth0, depth1 ), max( depth2, depth3 ) ); + + const lpfloat depthRangeScaleFactor = 0.75; // found empirically :) +#if XE_GTAO_USE_DEFAULT_CONSTANTS != 0 + const lpfloat effectRadius = depthRangeScaleFactor * (lpfloat)consts.EffectRadius * (lpfloat)XE_GTAO_DEFAULT_RADIUS_MULTIPLIER; + const lpfloat falloffRange = (lpfloat)XE_GTAO_DEFAULT_FALLOFF_RANGE * effectRadius; +#else + const lpfloat effectRadius = depthRangeScaleFactor * (lpfloat)consts.EffectRadius * (lpfloat)consts.RadiusMultiplier; + const lpfloat falloffRange = (lpfloat)consts.EffectFalloffRange * effectRadius; +#endif + const lpfloat falloffFrom = effectRadius * ((lpfloat)1-(lpfloat)consts.EffectFalloffRange); + // fadeout precompute optimisation + const lpfloat falloffMul = (lpfloat)-1.0 / ( falloffRange ); + const lpfloat falloffAdd = falloffFrom / ( falloffRange ) + (lpfloat)1.0; + + lpfloat weight0 = saturate( (maxDepth-depth0) * falloffMul + falloffAdd ); + lpfloat weight1 = saturate( (maxDepth-depth1) * falloffMul + falloffAdd ); + lpfloat weight2 = saturate( (maxDepth-depth2) * falloffMul + falloffAdd ); + lpfloat weight3 = saturate( (maxDepth-depth3) * falloffMul + falloffAdd ); + + lpfloat weightSum = weight0 + weight1 + weight2 + weight3; + return (weight0 * depth0 + weight1 * depth1 + weight2 * depth2 + weight3 * depth3) / weightSum; +} + +// This is also a good place to do non-linear depth conversion for cases where one wants the 'radius' (effectively the threshold between near-field and far-field GI), +// is required to be non-linear (i.e. very large outdoors environments). +lpfloat XeGTAO_ClampDepth( float depth ) +{ +#ifdef XE_GTAO_USE_HALF_FLOAT_PRECISION + return (lpfloat)clamp( depth, 0.0, 65504.0 ); +#else + return clamp( depth, 0.0, 3.402823466e+38 ); +#endif +} + +groupshared lpfloat g_scratchDepths[8][8]; +void XeGTAO_PrefilterDepths16x16( uint2 dispatchThreadID /*: SV_DispatchThreadID*/, uint2 groupThreadID /*: SV_GroupThreadID*/, const GTAOConstants consts, Texture2D sourceNDCDepth, SamplerState depthSampler, RWTexture2D outDepth0, RWTexture2D outDepth1, RWTexture2D outDepth2, RWTexture2D outDepth3, RWTexture2D outDepth4 ) +{ + // MIP 0 + const uint2 baseCoord = dispatchThreadID; + const uint2 pixCoord = baseCoord * 2; + float4 depths4 = sourceNDCDepth.GatherRed( depthSampler, float2( pixCoord * consts.ViewportPixelSize ), int2(1,1) ); + lpfloat depth0 = XeGTAO_ClampDepth( XeGTAO_ScreenSpaceToViewSpaceDepth( depths4.w, consts ) ); + lpfloat depth1 = XeGTAO_ClampDepth( XeGTAO_ScreenSpaceToViewSpaceDepth( depths4.z, consts ) ); + lpfloat depth2 = XeGTAO_ClampDepth( XeGTAO_ScreenSpaceToViewSpaceDepth( depths4.x, consts ) ); + lpfloat depth3 = XeGTAO_ClampDepth( XeGTAO_ScreenSpaceToViewSpaceDepth( depths4.y, consts ) ); + outDepth0[ pixCoord + uint2(0, 0) ] = (lpfloat)depth0; + outDepth0[ pixCoord + uint2(1, 0) ] = (lpfloat)depth1; + outDepth0[ pixCoord + uint2(0, 1) ] = (lpfloat)depth2; + outDepth0[ pixCoord + uint2(1, 1) ] = (lpfloat)depth3; + + // MIP 1 + lpfloat dm1 = XeGTAO_DepthMIPFilter( depth0, depth1, depth2, depth3, consts ); + outDepth1[ baseCoord ] = (lpfloat)dm1; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm1; + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 2 + [branch] + if( all( ( groupThreadID.xy % uint2(2, 2) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+1][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+1]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+1][groupThreadID.y+1]; + + lpfloat dm2 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth2[ baseCoord / 2 ] = (lpfloat)dm2; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm2; + } + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 3 + [branch] + if( all( ( groupThreadID.xy % uint2(4, 4) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+2][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+2]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+2][groupThreadID.y+2]; + + lpfloat dm3 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth3[ baseCoord / 4 ] = (lpfloat)dm3; + g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm3; + } + + GroupMemoryBarrierWithGroupSync( ); + + // MIP 4 + [branch] + if( all( ( groupThreadID.xy % uint2(8, 8) ) == uint2(0, 0) ) ) + { + lpfloat inTL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+0]; + lpfloat inTR = g_scratchDepths[groupThreadID.x+4][groupThreadID.y+0]; + lpfloat inBL = g_scratchDepths[groupThreadID.x+0][groupThreadID.y+4]; + lpfloat inBR = g_scratchDepths[groupThreadID.x+4][groupThreadID.y+4]; + + lpfloat dm4 = XeGTAO_DepthMIPFilter( inTL, inTR, inBL, inBR, consts ); + outDepth4[ baseCoord / 8 ] = (lpfloat)dm4; + //g_scratchDepths[ groupThreadID.x ][ groupThreadID.y ] = dm4; + } +} + +lpfloat4 XeGTAO_UnpackEdges( lpfloat _packedVal ) +{ + uint packedVal = (uint)(_packedVal * 255.5); + lpfloat4 edgesLRTB; + edgesLRTB.x = lpfloat((packedVal >> 6) & 0x03) / 3.0; // there's really no need for mask (as it's an 8 bit input) but I'll leave it in so it doesn't cause any trouble in the future + edgesLRTB.y = lpfloat((packedVal >> 4) & 0x03) / 3.0; + edgesLRTB.z = lpfloat((packedVal >> 2) & 0x03) / 3.0; + edgesLRTB.w = lpfloat((packedVal >> 0) & 0x03) / 3.0; + + return saturate( edgesLRTB ); +} + +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS +typedef lpfloat4 AOTermType; // .xyz is bent normal, .w is visibility term +#else +typedef lpfloat AOTermType; // .x is visibility term +#endif + +void XeGTAO_AddSample( AOTermType ssaoValue, lpfloat edgeValue, inout AOTermType sum, inout lpfloat sumWeight ) +{ + lpfloat weight = edgeValue; + + sum += (weight * ssaoValue); + sumWeight += weight; +} + +void XeGTAO_Output( uint2 pixCoord, RWTexture2D outputTexture, AOTermType outputValue, const uniform bool finalApply ) +{ +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + lpfloat visibility = outputValue.w * ((finalApply)?((lpfloat)XE_GTAO_OCCLUSION_TERM_SCALE):(1)); + lpfloat3 bentNormal = normalize(outputValue.xyz); + outputTexture[pixCoord.xy] = XeGTAO_EncodeVisibilityBentNormal( visibility, bentNormal ); +#else + outputValue *= (finalApply)?((lpfloat)XE_GTAO_OCCLUSION_TERM_SCALE):(1); + // OGSR: clamp before narrowing to R8_UINT; final visibility may overshoot 1. + outputTexture[pixCoord.xy] = uint(saturate(outputValue) * 255.0 + 0.5); +#endif +} + +void XeGTAO_DecodeGatherPartial( const uint4 packedValue, out AOTermType outDecoded[4] ) +{ + for( int i = 0; i < 4; i++ ) +#ifdef XE_GTAO_COMPUTE_BENT_NORMALS + XeGTAO_DecodeVisibilityBentNormal( packedValue[i], outDecoded[i].w, outDecoded[i].xyz ); +#else + outDecoded[i] = lpfloat(packedValue[i]) / lpfloat(255.0); +#endif +} + +void XeGTAO_Denoise( const uint2 pixCoordBase, const GTAOConstants consts, Texture2D sourceAOTerm, Texture2D sourceEdges, SamplerState texSampler, RWTexture2D outputTexture, const uniform bool finalApply ) +{ + const lpfloat blurAmount = (finalApply)?((lpfloat)consts.DenoiseBlurBeta):((lpfloat)consts.DenoiseBlurBeta/(lpfloat)5.0); + const lpfloat diagWeight = 0.85 * 0.5; + + AOTermType aoTerm[2]; // pixel pixCoordBase and pixel pixCoordBase + int2( 1, 0 ) + lpfloat4 edgesC_LRTB[2]; + lpfloat weightTL[2]; + lpfloat weightTR[2]; + lpfloat weightBL[2]; + lpfloat weightBR[2]; + + // gather edge and visibility quads, used later + const float2 gatherCenter = float2( pixCoordBase.x, pixCoordBase.y ) * consts.ViewportPixelSize; + lpfloat4 edgesQ0 = sourceEdges.GatherRed( texSampler, gatherCenter, int2( 0, 0 ) ); + lpfloat4 edgesQ1 = sourceEdges.GatherRed( texSampler, gatherCenter, int2( 2, 0 ) ); + lpfloat4 edgesQ2 = sourceEdges.GatherRed( texSampler, gatherCenter, int2( 1, 2 ) ); + + AOTermType visQ0[4]; XeGTAO_DecodeGatherPartial( sourceAOTerm.GatherRed( texSampler, gatherCenter, int2( 0, 0 ) ), visQ0 ); + AOTermType visQ1[4]; XeGTAO_DecodeGatherPartial( sourceAOTerm.GatherRed( texSampler, gatherCenter, int2( 2, 0 ) ), visQ1 ); + AOTermType visQ2[4]; XeGTAO_DecodeGatherPartial( sourceAOTerm.GatherRed( texSampler, gatherCenter, int2( 0, 2 ) ), visQ2 ); + AOTermType visQ3[4]; XeGTAO_DecodeGatherPartial( sourceAOTerm.GatherRed( texSampler, gatherCenter, int2( 2, 2 ) ), visQ3 ); + + for( int side = 0; side < 2; side++ ) + { + const int2 pixCoord = int2( pixCoordBase.x + side, pixCoordBase.y ); + + lpfloat4 edgesL_LRTB = XeGTAO_UnpackEdges( (side==0)?(edgesQ0.x):(edgesQ0.y) ); + lpfloat4 edgesT_LRTB = XeGTAO_UnpackEdges( (side==0)?(edgesQ0.z):(edgesQ1.w) ); + lpfloat4 edgesR_LRTB = XeGTAO_UnpackEdges( (side==0)?(edgesQ1.x):(edgesQ1.y) ); + lpfloat4 edgesB_LRTB = XeGTAO_UnpackEdges( (side==0)?(edgesQ2.w):(edgesQ2.z) ); + + edgesC_LRTB[side] = XeGTAO_UnpackEdges( (side==0)?(edgesQ0.y):(edgesQ1.x) ); + + // Edges aren't perfectly symmetrical: edge detection algorithm does not guarantee that a left edge on the right pixel will match the right edge on the left pixel (although + // they will match in majority of cases). This line further enforces the symmetricity, creating a slightly sharper blur. Works real nice with TAA. + edgesC_LRTB[side] *= lpfloat4( edgesL_LRTB.y, edgesR_LRTB.x, edgesT_LRTB.w, edgesB_LRTB.z ); + +#if 1 // this allows some small amount of AO leaking from neighbours if there are 3 or 4 edges; this reduces both spatial and temporal aliasing + const lpfloat leak_threshold = 2.5; const lpfloat leak_strength = 0.5; + lpfloat edginess = (saturate(4.0 - leak_threshold - dot( edgesC_LRTB[side], float4(1.0, 1.0, 1.0, 1.0) )) / (4-leak_threshold)) * leak_strength; + edgesC_LRTB[side] = saturate( edgesC_LRTB[side] + edginess ); +#endif + +#ifdef XE_GTAO_SHOW_EDGES + g_outputDbgImage[pixCoord] = 1.0 - lpfloat4( edgesC_LRTB[side].x, edgesC_LRTB[side].y * 0.5 + edgesC_LRTB[side].w * 0.5, edgesC_LRTB[side].z, 1.0 ); + //g_outputDbgImage[pixCoord] = 1 - float4( edgesC_LRTB[side].z, edgesC_LRTB[side].w , 1, 0 ); + //g_outputDbgImage[pixCoord] = edginess.xxxx; +#endif + + // for diagonals; used by first and second pass + weightTL[side] = diagWeight * (edgesC_LRTB[side].x * edgesL_LRTB.z + edgesC_LRTB[side].z * edgesT_LRTB.x); + weightTR[side] = diagWeight * (edgesC_LRTB[side].z * edgesT_LRTB.y + edgesC_LRTB[side].y * edgesR_LRTB.z); + weightBL[side] = diagWeight * (edgesC_LRTB[side].w * edgesB_LRTB.x + edgesC_LRTB[side].x * edgesL_LRTB.w); + weightBR[side] = diagWeight * (edgesC_LRTB[side].y * edgesR_LRTB.w + edgesC_LRTB[side].w * edgesB_LRTB.y); + + // first pass + AOTermType ssaoValue = (side==0)?(visQ0[1]):(visQ1[0]); + AOTermType ssaoValueL = (side==0)?(visQ0[0]):(visQ0[1]); + AOTermType ssaoValueT = (side==0)?(visQ0[2]):(visQ1[3]); + AOTermType ssaoValueR = (side==0)?(visQ1[0]):(visQ1[1]); + AOTermType ssaoValueB = (side==0)?(visQ2[2]):(visQ3[3]); + AOTermType ssaoValueTL = (side==0)?(visQ0[3]):(visQ0[2]); + AOTermType ssaoValueBR = (side==0)?(visQ3[3]):(visQ3[2]); + AOTermType ssaoValueTR = (side==0)?(visQ1[3]):(visQ1[2]); + AOTermType ssaoValueBL = (side==0)?(visQ2[3]):(visQ2[2]); + + lpfloat sumWeight = blurAmount; + AOTermType sum = ssaoValue * sumWeight; + + XeGTAO_AddSample( ssaoValueL, edgesC_LRTB[side].x, sum, sumWeight ); + XeGTAO_AddSample( ssaoValueR, edgesC_LRTB[side].y, sum, sumWeight ); + XeGTAO_AddSample( ssaoValueT, edgesC_LRTB[side].z, sum, sumWeight ); + XeGTAO_AddSample( ssaoValueB, edgesC_LRTB[side].w, sum, sumWeight ); + + XeGTAO_AddSample( ssaoValueTL, weightTL[side], sum, sumWeight ); + XeGTAO_AddSample( ssaoValueTR, weightTR[side], sum, sumWeight ); + XeGTAO_AddSample( ssaoValueBL, weightBL[side], sum, sumWeight ); + XeGTAO_AddSample( ssaoValueBR, weightBR[side], sum, sumWeight ); + + aoTerm[side] = sum / sumWeight; + + XeGTAO_Output( pixCoord, outputTexture, aoTerm[side], finalApply ); + +#ifdef XE_GTAO_SHOW_BENT_NORMALS + if( finalApply ) + { + g_outputDbgImage[pixCoord] = float4( DisplayNormalSRGB( aoTerm[side].xyz /** aoTerm[side].www*/ ), 1 ); + } +#endif + + } +} + + +// Generic viewspace normal generate pass +float3 XeGTAO_ComputeViewspaceNormal( const uint2 pixCoord, const GTAOConstants consts, Texture2D sourceNDCDepth, SamplerState depthSampler ) +{ + float2 normalizedScreenPos = (pixCoord + float2(0.5, 0.5)) * consts.ViewportPixelSize; + + float4 valuesUL = sourceNDCDepth.GatherRed( depthSampler, float2( pixCoord * consts.ViewportPixelSize ) ); + float4 valuesBR = sourceNDCDepth.GatherRed( depthSampler, float2( pixCoord * consts.ViewportPixelSize ), int2( 1, 1 ) ); + + // viewspace Z at the center + float viewspaceZ = XeGTAO_ScreenSpaceToViewSpaceDepth( valuesUL.y, consts ); //sourceViewspaceDepth.SampleLevel( depthSampler, normalizedScreenPos, 0 ).x; + + // viewspace Zs left top right bottom + const float pixLZ = XeGTAO_ScreenSpaceToViewSpaceDepth( valuesUL.x, consts ); + const float pixTZ = XeGTAO_ScreenSpaceToViewSpaceDepth( valuesUL.z, consts ); + const float pixRZ = XeGTAO_ScreenSpaceToViewSpaceDepth( valuesBR.z, consts ); + const float pixBZ = XeGTAO_ScreenSpaceToViewSpaceDepth( valuesBR.x, consts ); + + lpfloat4 edgesLRTB = XeGTAO_CalculateEdges( (lpfloat)viewspaceZ, (lpfloat)pixLZ, (lpfloat)pixRZ, (lpfloat)pixTZ, (lpfloat)pixBZ ); + + float3 CENTER = XeGTAO_ComputeViewspacePosition( normalizedScreenPos, viewspaceZ, consts ); + float3 LEFT = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2(-1, 0) * consts.ViewportPixelSize, pixLZ, consts ); + float3 RIGHT = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 1, 0) * consts.ViewportPixelSize, pixRZ, consts ); + float3 TOP = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 0, -1) * consts.ViewportPixelSize, pixTZ, consts ); + float3 BOTTOM = XeGTAO_ComputeViewspacePosition( normalizedScreenPos + float2( 0, 1) * consts.ViewportPixelSize, pixBZ, consts ); + return XeGTAO_CalculateNormal( edgesLRTB, CENTER, LEFT, RIGHT, TOP, BOTTOM ); +} diff --git a/ogsr_engine/Layers/xrRender/RenderTargetPhaseAA.cpp b/ogsr_engine/Layers/xrRender/RenderTargetPhaseAA.cpp index 46ad920de8..37a8592c51 100644 --- a/ogsr_engine/Layers/xrRender/RenderTargetPhaseAA.cpp +++ b/ogsr_engine/Layers/xrRender/RenderTargetPhaseAA.cpp @@ -445,8 +445,11 @@ bool CRenderTarget::ProcessDLSS() return true; } -void CRenderTarget::BeginPostprocess(CBackend& cmd_list, const bool temporalOutput) +void CRenderTarget::BeginPostprocess(CBackend& cmd_list, const bool temporalOutput, const bool skip_temporal_copy) { + m_pp_pingponged = false; + m_pp_current_is_combine = false; + // The last scene target is physically render-sized. Switch both the cached // target dimensions and the D3D viewport before any display-sized pass. u_setrt(cmd_list, GetDisplayWidth(), GetDisplayHeight(), nullptr, nullptr, nullptr, nullptr); @@ -454,10 +457,20 @@ void CRenderTarget::BeginPostprocess(CBackend& cmd_list, const bool temporalOutp if (temporalOutput) { + // DLSS/FSR already wrote combine. CAS (if any) samples that and writes postprocess0. + if (skip_temporal_copy) + { + m_pp_current_is_combine = true; + m_pp_remap_enabled = true; + return; + } + + PIX_EVENT(copy_pp_after_upscale); HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); } else if (GetRenderWidth() == GetDisplayWidth() && GetRenderHeight() == GetDisplayHeight()) { + PIX_EVENT(copy_pp_from_generic0); HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_0->pSurface); } else @@ -466,11 +479,13 @@ void CRenderTarget::BeginPostprocess(CBackend& cmd_list, const bool temporalOutp // stretch the render-sized scene so failure remains full-screen. RenderScreenTriangle(cmd_list, rt_Postprocess_0, s_temporal_resolve->E[0]); } + + m_pp_remap_enabled = true; } //***************************************************************************************************** -void CRenderTarget::ProcessCAS(CBackend& cmd_list) +void CRenderTarget::ProcessCAS(CBackend& cmd_list, const bool read_combine) { if (fis_zero(ps_r_cas)) return; @@ -478,8 +493,10 @@ void CRenderTarget::ProcessCAS(CBackend& cmd_list) PIX_EVENT(CAS); const Fvector4 params{std::max(ps_r_cas, 0.01f), 0.f, 0.f, 0.f}; - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_cas->E[0], [&]() { cmd_list.set_c("f_cas_intensity", params); }); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + // After skip_temporal_copy, current is combine (DLSS/FSR). Element 1 samples that. + // After a combine→postprocess0 copy, current is postprocess0. Element 0 samples that. + RenderScreenTriangle(cmd_list, pp_dst(), read_combine ? s_cas->E[1] : s_cas->E[0], [&]() { cmd_list.set_c("f_cas_intensity", params); }); + pp_flip(); } //***************************************************************************************************** @@ -823,10 +840,14 @@ void CRenderTarget::PhaseAA(CBackend& cmd_list) EndTemporalUpscaleInput(); RImplementation.rmNormal(cmd_list); - BeginPostprocess(cmd_list, temporalOutput); + + const bool cas = !fis_zero(ps_r_cas) && ps_r_pp_aa_mode != SMAA; + const bool cas_from_combine = temporalOutput && cas && s_cas->E[1]; + + BeginPostprocess(cmd_list, temporalOutput, cas_from_combine); if (ps_r_pp_aa_mode != SMAA) - ProcessCAS(cmd_list); + ProcessCAS(cmd_list, cas_from_combine); } //***************************************************************************************************** diff --git a/ogsr_engine/Layers/xrRender/RenderTargetPhaseRainDrops.cpp b/ogsr_engine/Layers/xrRender/RenderTargetPhaseRainDrops.cpp index d5bdde857b..cf6eb6b5fb 100644 --- a/ogsr_engine/Layers/xrRender/RenderTargetPhaseRainDrops.cpp +++ b/ogsr_engine/Layers/xrRender/RenderTargetPhaseRainDrops.cpp @@ -78,7 +78,6 @@ void CRenderTarget::phase_rain_drops(CBackend& cmd_list) PIX_EVENT(phase_rain_drops); const Fvector4 params{rain_drops_factor, ps_r2_rain_drops_intensity, ps_r2_rain_drops_speed, 0.0f}; - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_rain_drops->E[0], [&]() { cmd_list.set_c("rain_drops_params", params); }); - - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_rain_drops->E[0], [&]() { cmd_list.set_c("rain_drops_params", params); }); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/RenderTargetPhaseSSSS.cpp b/ogsr_engine/Layers/xrRender/RenderTargetPhaseSSSS.cpp index 9fff22f659..0b902b39f9 100644 --- a/ogsr_engine/Layers/xrRender/RenderTargetPhaseSSSS.cpp +++ b/ogsr_engine/Layers/xrRender/RenderTargetPhaseSSSS.cpp @@ -33,9 +33,8 @@ void CRenderTarget::PhaseSSSS(CBackend& cmd_list) [&]() { cmd_list.set_c("ssss_params", ps_r_prop_ss_sample_step_phase1, ps_r_prop_ss_radius, 0.0f, 0.0f); }); // Combine - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_ssss_mrmnwar->E[4], [&]() { cmd_list.set_c("ssss_params", intensity, ps_r_prop_ss_blend, 0.0f, 0.0f); }); - - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_ssss_mrmnwar->E[4], [&]() { cmd_list.set_c("ssss_params", intensity, ps_r_prop_ss_blend, 0.0f, 0.0f); }); + pp_flip(); } else if (mode == SS_SS_OGSE) { @@ -62,9 +61,8 @@ void CRenderTarget::PhaseSSSS(CBackend& cmd_list) //***BLEND PASS*** // Combining sunshafts texture and image for further processing - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_ssss_ogse->E[4], + RenderScreenTriangle(cmd_list, pp_dst(), s_ssss_ogse->E[4], [&]() { cmd_list.set_c("ssss_params", intensity, ps_r_ss_sunshafts_length, 0.0f, ps_r_ss_sunshafts_radius); }); - - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + pp_flip(); } } diff --git a/ogsr_engine/Layers/xrRender/RenderTargetRenderScreenQuad.cpp b/ogsr_engine/Layers/xrRender/RenderTargetRenderScreenQuad.cpp index 455a496078..bac999de36 100644 --- a/ogsr_engine/Layers/xrRender/RenderTargetRenderScreenQuad.cpp +++ b/ogsr_engine/Layers/xrRender/RenderTargetRenderScreenQuad.cpp @@ -1,5 +1,38 @@ #include "stdafx.h" +void CRenderTarget::pp_remap_scene_srv(CBackend& cmd_list, ShaderElement* se) const +{ + if (!m_pp_remap_enabled || !se) + return; + + CTexture* const current = pp_src()->pTexture._get(); + CTexture* const pp0 = rt_Postprocess_0->pTexture._get(); + CTexture* const combine = rt_Generic_combine->pTexture._get(); + if (!current || !pp0 || !combine) + return; + + for (u32 p = 0; p < se->passes.size(); ++p) + { + SPass* pass = se->passes[p]._get(); + if (!pass || !pass->T) + continue; + + for (const auto& loader : *pass->T) + { + if (loader.first >= CTexture::rstVertex) + continue; + + CTexture* tex = loader.second._get(); + if (!tex || tex == current || (tex != pp0 && tex != combine)) + continue; + if (!current->bind) + continue; + + current->bind(cmd_list, loader.first); + } + } +} + void CRenderTarget::RenderScreenTriangle(CBackend& cmd_list, const ref_rt& rt, ref_selement& sh, const std::function& lambda) { u_setrt(cmd_list, rt->dwWidth, rt->dwHeight, rt->pRT, nullptr, nullptr, nullptr); @@ -9,6 +42,7 @@ void CRenderTarget::RenderScreenTriangle(CBackend& cmd_list, const ref_rt& rt, r cmd_list.set_Stencil(FALSE); cmd_list.set_Element(sh); + pp_remap_scene_srv(cmd_list, sh._get()); lambda(); @@ -45,6 +79,7 @@ void CRenderTarget::RenderScreenQuad(CBackend& cmd_list, const u32 w, const u32 RImplementation.Vertex.Unlock(4, g_combine->vb_stride); cmd_list.set_Element(sh); + pp_remap_scene_srv(cmd_list, sh._get()); lambda(); diff --git a/ogsr_engine/Layers/xrRender/rendertarget_phase_dof.cpp b/ogsr_engine/Layers/xrRender/rendertarget_phase_dof.cpp index 52093b9ffc..1ef42de543 100644 --- a/ogsr_engine/Layers/xrRender/rendertarget_phase_dof.cpp +++ b/ogsr_engine/Layers/xrRender/rendertarget_phase_dof.cpp @@ -9,8 +9,6 @@ void CRenderTarget::phase_dof(CBackend& cmd_list) PIX_EVENT(phase_DOF); RenderScreenTriangle(cmd_list, rt_dof, s_dof->E[0]); - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_dof->E[1]); - - //Resolve RT - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_dof->E[1]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/rendertarget_phase_fakescope.cpp b/ogsr_engine/Layers/xrRender/rendertarget_phase_fakescope.cpp index 60bc1d35c5..2ad4e62e81 100644 --- a/ogsr_engine/Layers/xrRender/rendertarget_phase_fakescope.cpp +++ b/ogsr_engine/Layers/xrRender/rendertarget_phase_fakescope.cpp @@ -6,7 +6,6 @@ void CRenderTarget::phase_fakescope(CBackend& cmd_list) if (Device.IsAltScopeActive()) return; - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_fakescope->E[0]); - - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_fakescope->E[0]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/rendertarget_phase_gasmask_dudv.cpp b/ogsr_engine/Layers/xrRender/rendertarget_phase_gasmask_dudv.cpp index e287776828..98b0a1d871 100644 --- a/ogsr_engine/Layers/xrRender/rendertarget_phase_gasmask_dudv.cpp +++ b/ogsr_engine/Layers/xrRender/rendertarget_phase_gasmask_dudv.cpp @@ -7,14 +7,14 @@ void CRenderTarget::phase_gasmask_dudv(CBackend& cmd_list) const Fvector4 params{0.f, 0.f, static_cast(ps_r2_ls_flags_ext.test(R2FLAGEXT_VISOR_REFL_CONTROL) && ps_r2_ls_flags_ext.test(R2FLAGEXT_VISOR_REFL)), static_cast(ps_r2_ls_flags_ext.test(R2FLAGEXT_MASK_CONTROL) && ps_r2_ls_flags_ext.test(R2FLAGEXT_MASK))}; - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_gasmask_dudv->E[0], [&]() { + RenderScreenTriangle(cmd_list, pp_dst(), s_gasmask_dudv->E[0], [&]() { cmd_list.set_c("mask_control", params); cmd_list.set_c("addon_VControl", ps_r2_visor_refl_intensity, ps_r2_visor_refl_radius, 0.f, 1.f); }); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + pp_flip(); RenderScreenTriangle(cmd_list, rt_mask_drops_blur, s_gasmask_dudv->E[1]); - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_gasmask_dudv->E[2]); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_gasmask_dudv->E[2]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/rendertarget_phase_nightvision.cpp b/ogsr_engine/Layers/xrRender/rendertarget_phase_nightvision.cpp index a5b4409b41..511c71df77 100644 --- a/ogsr_engine/Layers/xrRender/rendertarget_phase_nightvision.cpp +++ b/ogsr_engine/Layers/xrRender/rendertarget_phase_nightvision.cpp @@ -4,6 +4,6 @@ void CRenderTarget::phase_nightvision(CBackend& cmd_list) { PIX_EVENT(phase_nightvision); - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_nightvision->E[0]); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_nightvision->E[0]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/rendertarget_phase_thermalvision.cpp b/ogsr_engine/Layers/xrRender/rendertarget_phase_thermalvision.cpp index 7280253493..5fd93d1cfc 100644 --- a/ogsr_engine/Layers/xrRender/rendertarget_phase_thermalvision.cpp +++ b/ogsr_engine/Layers/xrRender/rendertarget_phase_thermalvision.cpp @@ -12,6 +12,6 @@ void CRenderTarget::phase_heatvision_overlay(CBackend& cmd_list) { PIX_EVENT(phase_heatvision_overlay); - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_heatvision->E[1]); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_heatvision->E[1]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRender/xrRender_console.cpp b/ogsr_engine/Layers/xrRender/xrRender_console.cpp index 48a356014a..b561929180 100644 --- a/ogsr_engine/Layers/xrRender/xrRender_console.cpp +++ b/ogsr_engine/Layers/xrRender/xrRender_console.cpp @@ -100,7 +100,12 @@ u32 ps_preset = 2; constexpr xr_token qpreset_token[] = {{"Minimum", 0}, {"Low", 1}, {"Default", 2}, {"High", 3}, {"Extreme", 4}, {nullptr, 0}}; u32 ps_r_ao_mode = AO_MODE_SSDO; -constexpr xr_token ao_mode_token[] = {{"st_gtao", AO_MODE_GTAO}, {"st_ssdo", AO_MODE_SSDO}, {nullptr, 0}}; +constexpr xr_token ao_mode_token[] = {{"st_gtao", AO_MODE_GTAO}, {"st_ssdo", AO_MODE_SSDO}, {"st_xegtao", AO_MODE_XEGTAO}, {nullptr, 0}}; +float ps_r_xegtao_radius = 0.5f; +BOOL ps_r_xegtao_bent_normals = FALSE; + +u32 ps_r_ao_resolution = AO_RES_FULL; +constexpr xr_token ao_resolution_token[] = {{"legacy", AO_RES_LEGACY}, {"full", AO_RES_FULL}, {"half", AO_RES_HALF}, {nullptr, 0}}; u32 ps_r_ao_quality = 0; constexpr xr_token qssao_token[] = {{"st_opt_off", 0}, @@ -825,7 +830,10 @@ void xrRender_initconsole() CMD4(CCC_Float, "r_SunShafts_Blend", &ps_r_prop_ss_blend, 0.01f, 1.0f); CMD3(CCC_Token, "r_ao_mode", &ps_r_ao_mode, ao_mode_token); + CMD4(CCC_Float, "r_xegtao_radius", &ps_r_xegtao_radius, 0.05f, 4.0f); + CMD2(CCC_Bool, "r_xegtao_bent_normals", &ps_r_xegtao_bent_normals); // Requires vid_restart. CMD3(CCC_Token, "r2_ssao", &ps_r_ao_quality, qssao_token); + CMD3(CCC_Token, "r_ao_resolution", &ps_r_ao_resolution, ao_resolution_token); CMD3(CCC_Mask64, "r4_enable_tessellation", &ps_r2_ls_flags_ext, R2FLAGEXT_ENABLE_TESSELLATION); // Need restart diff --git a/ogsr_engine/Layers/xrRender/xrRender_console.h b/ogsr_engine/Layers/xrRender/xrRender_console.h index 3e16fe766a..d6bea34dc5 100644 --- a/ogsr_engine/Layers/xrRender/xrRender_console.h +++ b/ogsr_engine/Layers/xrRender/xrRender_console.h @@ -48,9 +48,20 @@ extern ECORE_API u32 ps_r_ao_quality; // = 0; enum : u32 { AO_MODE_GTAO, - AO_MODE_SSDO + AO_MODE_SSDO, + AO_MODE_XEGTAO }; extern ECORE_API u32 ps_r_ao_mode; +extern ECORE_API float ps_r_xegtao_radius; +extern ECORE_API BOOL ps_r_xegtao_bent_normals; // Requires vid_restart. + +enum : u32 +{ + AO_RES_LEGACY, + AO_RES_FULL, + AO_RES_HALF +}; +extern ECORE_API u32 ps_r_ao_resolution; extern ECORE_API u32 ps_r_sun_quality; // = 0; diff --git a/ogsr_engine/Layers/xrRenderDX10/StateManager/dx10ShaderResourceStateCache.cpp b/ogsr_engine/Layers/xrRenderDX10/StateManager/dx10ShaderResourceStateCache.cpp index 45d7eed4df..f7933381c3 100644 --- a/ogsr_engine/Layers/xrRenderDX10/StateManager/dx10ShaderResourceStateCache.cpp +++ b/ogsr_engine/Layers/xrRenderDX10/StateManager/dx10ShaderResourceStateCache.cpp @@ -10,6 +10,7 @@ void dx10ShaderResourceStateCache::ResetDeviceState() ZeroMemory(m_VSViews, sizeof(m_VSViews)); ZeroMemory(m_HSViews, sizeof(m_HSViews)); ZeroMemory(m_DSViews, sizeof(m_DSViews)); + ZeroMemory(m_CSViews, sizeof(m_CSViews)); m_uiMinPSView = 0xFFFFFFFF; m_uiMaxPSView = 0xFFFFFFFF; @@ -26,11 +27,15 @@ void dx10ShaderResourceStateCache::ResetDeviceState() m_uiMinDSView = 0xFFFFFFFF; m_uiMaxDSView = 0xFFFFFFFF; + m_uiMinCSView = 0xFFFFFFFF; + m_uiMaxCSView = 0xFFFFFFFF; + m_bUpdatePSViews = false; m_bUpdateGSViews = false; m_bUpdateVSViews = false; m_bUpdateDSViews = false; m_bUpdateHSViews = false; + m_bUpdateCSViews = false; } void dx10ShaderResourceStateCache::Apply(u32 context_id) @@ -208,4 +213,4 @@ void dx10ShaderResourceStateCache::SetCSResource(u32 uiSlot, ID3DShaderResourceV m_uiMaxCSView = uiSlot; } } -} \ No newline at end of file +} diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4.cpp index d5aef74031..35b827cc3b 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4.cpp @@ -718,6 +718,8 @@ HRESULT CRender::shader_compile(LPCSTR name, DWORD const* pSrcData, UINT SrcData appendShaderOption(ps_r2_ls_flags_ext.test(R2FLAGEXT_MOTION_BLUR), "USE_MBLUR", "1"); appendShaderOption(ps_r_ao_mode == AO_MODE_GTAO, "USE_GTAO", "1"); + appendShaderOption(ps_r_ao_mode == AO_MODE_XEGTAO && ps_r_ao_quality && ps_r_xegtao_bent_normals, + "USE_XEGTAO_BENT_NORMALS", "1"); if (ps_r_ao_quality) { @@ -1077,4 +1079,4 @@ void CRender::End() TracyD3D11Collect(HW.profiler_ctx); Target->reset_target_dimensions(); -} \ No newline at end of file +} diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp index 6442f755e3..5fbafcc2d9 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_R_render.cpp @@ -389,6 +389,10 @@ void CRender::Render() LP_normal.vis_prepare(cmd_list); } + auto copy_scene_depth = [&](const ref_rt& dest) { + HW.get_context(cmd_list.context_id)->CopyResource(dest->pSurface, Target->rt_Base_Depth->pSurface); + }; + //******* Main render :: PART-1 (second) { PIX_EVENT(DEFER_PART1_SPLIT); @@ -398,35 +402,30 @@ void CRender::Render() dsgraph.r_dsgraph_render_lods(); if (Details) Details->Render(cmd_list); + + // Snapshot scene depth without HUD for 3D-scope z-write. + if (!dsgraph.mapScopeHUD.empty()) { - { - PIX_EVENT(copy_zbuffer_scope); + PIX_EVENT(copy_zbuffer_scope); + copy_scene_depth(Target->rt_tempzb); + } - ID3D11Resource* res{}; - Target->get_base_zb()->GetResource(&res); - HW.get_context(cmd_list.context_id)->CopyResource(Target->rt_tempzb->pSurface, res); - _RELEASE(res); - } - dsgraph.r_dsgraph_render_hud(); - { - PIX_EVENT(copy_zbuffer_scope_depth); + dsgraph.r_dsgraph_render_hud(); - ID3D11Resource* res{}; - Target->get_base_zb()->GetResource(&res); - HW.get_context(cmd_list.context_id)->CopyResource(Target->rt_tempzb_dof->pSurface, res); - _RELEASE(res); - } - dsgraph.r_dsgraph_render_hud_scope_depth(); + // Snapshot scene+HUD depth for DOF. Same enable check as phase_dof. + const auto& dof_params = shader_exports.get_dof_params(); + if (!(fis_zero(dof_params.x) && fis_zero(dof_params.y) && fis_zero(dof_params.z) && fis_zero(dof_params.w))) + { + PIX_EVENT(copy_zbuffer_scope_depth); + copy_scene_depth(Target->rt_tempzb_dof); } + + dsgraph.r_dsgraph_render_hud_scope_depth(); } { PIX_EVENT(copy_zbuffer); - - ID3D11Resource* res{}; - Target->get_base_zb()->GetResource(&res); - HW.get_context(cmd_list.context_id)->CopyResource(Target->rt_zbuffer->pSurface, res); - _RELEASE(res); + copy_scene_depth(Target->rt_zbuffer); } // Wall marks diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.cpp index c7006cde0d..aa09e66866 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.cpp @@ -189,6 +189,10 @@ CRenderTarget::CRenderTarget() SetTemporalRenderSize(Device.dwWidth, Device.dwHeight, Device.dwWidth, Device.dwHeight); ConfigureTemporalRenderSize(); + m_ao_enabled = ps_r_ao_quality != 0; + m_ao_mode = ps_r_ao_mode; + m_xegtao_bent_normals = m_ao_enabled && m_ao_mode == AO_MODE_XEGTAO && ps_r_xegtao_bent_normals; + param_blur = 0.f; param_gray = 0.f; param_noise = 0.f; @@ -225,6 +229,15 @@ CRenderTarget::CRenderTarget() rt_Position.create(r2_RT_P, w, h, DXGI_FORMAT_R16G16B16A16_FLOAT); + if (m_ao_enabled) + { + // XeGTAO already quantizes each component to 8 bits. Bent mode keeps + // visibility in R and the encoded view-space direction in GBA. + rt_ao.create("$user$ao", w, h, m_xegtao_bent_normals ? DXGI_FORMAT_R8G8B8A8_UNORM : DXGI_FORMAT_R16_FLOAT); + if (m_ao_mode != AO_MODE_XEGTAO) + rt_ao_half.create("$user$ao_half", (w + 1) / 2, (h + 1) / 2, DXGI_FORMAT_R16G16B16A16_FLOAT); + } + rt_Accumulator.create(r2_RT_accum, w, h, DXGI_FORMAT_R16G16B16A16_FLOAT); rt_Color.create(r2_RT_albedo, w, h, DXGI_FORMAT_R8G8B8A8_UNORM); @@ -327,6 +340,13 @@ CRenderTarget::CRenderTarget() s_flare.create("effects\\lensflare", "shaders\\lensflare"); s_lut.create("ogsr_lut"); s_ssr.create("ogsr_ssr"); + if (m_ao_enabled) + { + if (m_ao_mode == AO_MODE_XEGTAO) + InitXeGTAO(); + else + s_ao.create("ogsr_ao"); + } s_ssfx_bloom.create("ogsr_bloom"); s_ssfx_bloom_lens.create("ogsr_bloom_flares"); @@ -594,6 +614,7 @@ CRenderTarget::~CRenderTarget() DestroyDLSS(); DestroyFSR(); + DestroyXeGTAO(); _RELEASE(m_ImguiSRV); _RELEASE(m_ImguiTex); diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.h b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.h index e7637efbc4..2697b5c3ba 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.h +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget.h @@ -3,6 +3,8 @@ #include "../xrRender/ColorMapManager.h" class light; +struct ShaderElement; +class XeGTAOResources; static void dummy(){} @@ -17,6 +19,13 @@ class CRenderTarget : public IRender_Target u32 m_displayHeight{}; bool m_temporalUpscaleInput{}; bool m_resetTemporalHistory{true}; + bool m_ao_enabled{}; // Matches SSAO_QUALITY when the target's shaders were compiled. + u32 m_ao_mode{}; // Method changes, like quality changes, require vid_restart. + bool m_xegtao_bent_normals{}; // Latched with the AO shaders and texture formats. + XeGTAOResources* m_xegtao{}; + void InitXeGTAO(); + void DestroyXeGTAO(); + void phase_xegtao(CBackend& cmd_list); u32 dwAccumulatorClearMark; u32 dwFlareClearMark; @@ -43,6 +52,9 @@ class CRenderTarget : public IRender_Target ref_rt rt_Color; // 64/32bit,fat (r,g,b,specular-gloss) (or decompressed MET-8-8-8-8) ref_rt rt_Velocity; // r2_RT_velocity + ref_rt rt_ao; // Single-channel visibility at internal render resolution. + ref_rt rt_ao_half; // Half-size visibility, view depth and packed normal for reconstruction. + ref_rt rt_zbuffer; // r2_RT_zbuffer ref_rt rt_tempzb, rt_tempzb_dof; @@ -167,6 +179,7 @@ class CRenderTarget : public IRender_Target ref_geom g_combine_2UV; ref_geom g_combine_cuboid; ref_shader s_combine; + ref_shader s_ao; ref_shader s_combine_volumetric; ref_shader s_blur; @@ -277,6 +290,7 @@ class CRenderTarget : public IRender_Target void phase_ssfx_bloom(CBackend& cmd_list); void phase_luminance(CBackend& cmd_list); void phase_combine(CBackend& cmd_list); + void phase_ao(CBackend& cmd_list); void phase_pp(CBackend& cmd_list); void phase_combine_volumetric(CBackend& cmd_list); @@ -342,6 +356,21 @@ class CRenderTarget : public IRender_Target #endif private: + // Display-sized post: latest image is in combine or postprocess0. + bool m_pp_current_is_combine{}; + bool m_pp_remap_enabled{}; + bool m_pp_pingponged{}; + + ref_rt& pp_src() { return m_pp_current_is_combine ? rt_Generic_combine : rt_Postprocess_0; } + const ref_rt& pp_src() const { return m_pp_current_is_combine ? rt_Generic_combine : rt_Postprocess_0; } + ref_rt& pp_dst() { return m_pp_current_is_combine ? rt_Postprocess_0 : rt_Generic_combine; } + void pp_flip() + { + m_pp_current_is_combine = !m_pp_current_is_combine; + m_pp_pingponged = true; + } + void pp_remap_scene_srv(CBackend& cmd_list, ShaderElement* se) const; + void RenderScreenTriangle(CBackend& cmd_list, const ref_rt& rt, ref_selement& sh, const std::function& lambda = dummy); void RenderScreenQuad(CBackend& cmd_list, const u32 w, u32 const h, const ref_rt& rt, ref_selement& sh, const std::function& lambda = dummy); @@ -367,8 +396,8 @@ class CRenderTarget : public IRender_Target bool reset_3dss_rendertarget(const bool need_reset = false); - void ProcessCAS(CBackend& cmd_list); - void BeginPostprocess(CBackend& cmd_list, bool temporalOutput); + void ProcessCAS(CBackend& cmd_list, bool read_combine); + void BeginPostprocess(CBackend& cmd_list, bool temporalOutput, bool skip_temporal_copy = false); void PhaseSSSS(CBackend& cmd_list); diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_PP.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_PP.cpp index b103534544..c1a1497d61 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_PP.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_PP.cpp @@ -87,7 +87,9 @@ void CRenderTarget::phase_pp(CBackend& cmd_list) RImplementation.rmNormal(cmd_list); const bool bCMap = u_need_CM(); - cmd_list.set_Element(s_postprocess->E[bCMap ? 4 : 0]); + ref_selement& sh = s_postprocess->E[bCMap ? 4 : 0]; + cmd_list.set_Element(sh); + pp_remap_scene_srv(cmd_list, sh._get()); const int gblend = clampr(iFloor((1 - param_gray) * 255.f), 0, 255); const int nblend = clampr(iFloor((1 - param_noise) * 255.f), 0, 255); @@ -122,4 +124,5 @@ void CRenderTarget::phase_pp(CBackend& cmd_list) cmd_list.set_c("c_colormap", param_color_map_influence, param_color_map_interpolate, 0, 0); cmd_list.set_Geometry(g_postprocess); cmd_list.Render(D3DPT_TRIANGLELIST, Offset, 0, 4, 0, 2); + m_pp_remap_enabled = false; } diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_ao.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_ao.cpp new file mode 100644 index 0000000000..a9522fb6c3 --- /dev/null +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_ao.cpp @@ -0,0 +1,31 @@ +#include "stdafx.h" + +void CRenderTarget::phase_ao(CBackend& cmd_list) +{ + PIX_EVENT(phase_ao); + + if (m_ao_mode == AO_MODE_XEGTAO) + { + phase_xegtao(cmd_list); + return; + } + + cmd_list.set_ColorWriteEnable(); + + if (ps_r_ao_resolution == AO_RES_HALF) + { + { + PIX_EVENT(ao_evaluate_half); + RenderScreenTriangle(cmd_list, rt_ao_half, s_ao->E[1]); + } + { + PIX_EVENT(ao_resolve); + RenderScreenTriangle(cmd_list, rt_ao, s_ao->E[2]); + } + } + else + { + PIX_EVENT(ao_evaluate_full); + RenderScreenTriangle(cmd_list, rt_ao, s_ao->E[0]); + } +} diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_combine.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_combine.cpp index 321bfef136..d1724cf785 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_combine.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_combine.cpp @@ -9,6 +9,10 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) { ZoneScoped; + const bool separate_ao = m_ao_enabled && (m_ao_mode == AO_MODE_XEGTAO || ps_r_ao_resolution != AO_RES_LEGACY); + if (separate_ao) + phase_ao(cmd_list); + //*** exposure-pipeline { // if (t_LUM_src != rt_LUM_pool[0]->pTexture) @@ -18,6 +22,7 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) } u_setrt(cmd_list, rt_Generic_0, nullptr, nullptr, nullptr, rt_Base_Depth->pZRT[cmd_list.context_id]); + RImplementation.rmNormal(cmd_list); cmd_list.set_CullMode(CULL_NONE); cmd_list.set_Stencil(FALSE); @@ -75,7 +80,7 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) t_envmap_1->surface_set(e1); // Draw - cmd_list.set_Element(s_combine->E[0]); + cmd_list.set_Element(s_combine->E[separate_ao ? 4 : 0]); cmd_list.set_Geometry(TriangleGeom); cmd_list.set_c("m_inv_v", Device.mInvView); @@ -215,6 +220,13 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) { PIX_EVENT(phase_3DSSReticle); + // Reticle samples $user$generic_combine while drawing into postprocess0. + if (!m_pp_current_is_combine && m_pp_pingponged) + { + PIX_EVENT(copy_pp_3dss); + HW.get_context(cmd_list.context_id)->CopyResource(pp_dst()->pSurface, pp_src()->pSurface); + } + // The reticle is composited after temporal upscaling. Scene depth is // render-sized and cannot be bound with this display-sized color RT. u_setrt(cmd_list, rt_Postprocess_0, nullptr, nullptr, nullptr, nullptr); @@ -225,6 +237,7 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) cmd_list.set_ColorWriteEnable(); dsgraph.r_dsgraph_render_scope_sorted(upscaled_3dss); + m_pp_current_is_combine = false; } // Compute blur textures @@ -280,13 +293,12 @@ void CRenderTarget::phase_combine(CBackend& cmd_list) m_blur_scale.set(scale, -scale).div(12.f); } - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_combine->E[3], [&]() { + RenderScreenTriangle(cmd_list, pp_dst(), s_combine->E[3], [&]() { cmd_list.set_c("m_current", m_current); cmd_list.set_c("m_previous", m_previous); cmd_list.set_c("m_blur", m_blur_scale.x, m_blur_scale.y, 0, 0); }); - - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + pp_flip(); } { diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_lut.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_lut.cpp index 39e29ad2b1..1f2c360806 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_lut.cpp +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_lut.cpp @@ -7,6 +7,6 @@ void CRenderTarget::phase_lut(CBackend& cmd_list) PIX_EVENT(phase_LUT); - RenderScreenTriangle(cmd_list, rt_Generic_combine, s_lut->E[0]); - HW.get_context(cmd_list.context_id)->CopyResource(rt_Postprocess_0->pSurface, rt_Generic_combine->pSurface); + RenderScreenTriangle(cmd_list, pp_dst(), s_lut->E[0]); + pp_flip(); } diff --git a/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_xegtao.cpp b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_xegtao.cpp new file mode 100644 index 0000000000..027c7d3a5b --- /dev/null +++ b/ogsr_engine/Layers/xrRenderPC_R4/r4_rendertarget_phase_xegtao.cpp @@ -0,0 +1,173 @@ +#include "stdafx.h" +#include "../xrRender/dxRenderDeviceRender.h" +#include + +using Microsoft::WRL::ComPtr; +static_assert(sizeof(Fvector4) == 16, "XeGTAO constant-buffer layout must match HLSL float4"); + +// Kept local to AO: the general render-target abstraction has no mip UAVs. +class XeGTAOResources +{ +public: + struct Texture + { + ComPtr surface; + ComPtr srv; + ComPtr uav[5]; + + void create(u32 width, u32 height, DXGI_FORMAT format, u32 mips = 1) + { + D3D11_TEXTURE2D_DESC desc{}; + desc.Width = width; + desc.Height = height; + desc.MipLevels = mips; + desc.ArraySize = 1; + desc.Format = format; + desc.SampleDesc.Count = 1; + desc.Usage = D3D11_USAGE_DEFAULT; + desc.BindFlags = D3D11_BIND_SHADER_RESOURCE | D3D11_BIND_UNORDERED_ACCESS; + CHK_DX(HW.pDevice->CreateTexture2D(&desc, nullptr, surface.GetAddressOf())); + CHK_DX(HW.pDevice->CreateShaderResourceView(surface.Get(), nullptr, srv.GetAddressOf())); + for (u32 mip = 0; mip < mips; ++mip) + { + D3D11_UNORDERED_ACCESS_VIEW_DESC view{}; + view.Format = format; + view.ViewDimension = D3D11_UAV_DIMENSION_TEXTURE2D; + view.Texture2D.MipSlice = mip; + CHK_DX(HW.pDevice->CreateUnorderedAccessView(surface.Get(), &view, uav[mip].GetAddressOf())); + } + } + }; + + u32 width{}, height{}; + Texture depth, ao, edges, filtered; + ComPtr constants; + ComPtr pointClamp; + ref_cs prefilter, evaluate, denoise; + ref_texture output; + ref_shader exportAO; + + ~XeGTAOResources() + { + if (output) + output->surface_set(nullptr); + } +}; + +void CRenderTarget::InitXeGTAO() +{ + R_ASSERT(!m_xegtao); + m_xegtao = xr_new(); + auto& xe = *m_xegtao; + // Five mips and complete 16x16 prefilter groups, including odd/tiny sizes. + xe.width = (m_renderWidth + 15u) & ~15u; + xe.height = (m_renderHeight + 15u) & ~15u; + xe.depth.create(xe.width, xe.height, DXGI_FORMAT_R16_FLOAT, 5); + const DXGI_FORMAT aoFormat = m_xegtao_bent_normals ? DXGI_FORMAT_R32_UINT : DXGI_FORMAT_R8_UINT; + xe.ao.create(xe.width, xe.height, aoFormat); + xe.edges.create(xe.width, xe.height, DXGI_FORMAT_R8_UNORM); + xe.filtered.create(xe.width, xe.height, aoFormat); + + D3D11_BUFFER_DESC cb{}; + cb.ByteWidth = 3 * sizeof(Fvector4); // ogsr_xegtao_common.h: three float4s + cb.Usage = D3D11_USAGE_DYNAMIC; + cb.BindFlags = D3D11_BIND_CONSTANT_BUFFER; + cb.CPUAccessFlags = D3D11_CPU_ACCESS_WRITE; + CHK_DX(HW.pDevice->CreateBuffer(&cb, nullptr, xe.constants.GetAddressOf())); + + D3D11_SAMPLER_DESC sampler{}; + sampler.Filter = D3D11_FILTER_MIN_MAG_MIP_POINT; + sampler.AddressU = sampler.AddressV = sampler.AddressW = D3D11_TEXTURE_ADDRESS_CLAMP; + sampler.ComparisonFunc = D3D11_COMPARISON_NEVER; + sampler.MaxAnisotropy = 1; + sampler.MaxLOD = D3D11_FLOAT32_MAX; + CHK_DX(HW.pDevice->CreateSamplerState(&sampler, xe.pointClamp.GetAddressOf())); + + xe.prefilter = DEV->_CreateCS("ogsr_xegtao_prefilter"); + xe.evaluate = DEV->_CreateCS("ogsr_xegtao_main"); + xe.denoise = DEV->_CreateCS("ogsr_xegtao_denoise"); + xe.output.create("$user$xegtao"); + xe.output->surface_set(xe.filtered.surface.Get()); + xe.exportAO.create("ogsr_xegtao"); + + Msg("* XeGTAO: full internal resolution %ux%u (padded %ux%u), quality %u, bent normals %s; r_ao_resolution does not apply", + m_renderWidth, m_renderHeight, xe.width, xe.height, ps_r_ao_quality, m_xegtao_bent_normals ? "on" : "off"); +} + +void CRenderTarget::DestroyXeGTAO() +{ + xr_delete(m_xegtao); +} + +void CRenderTarget::phase_xegtao(CBackend& cmd_list) +{ + PIX_EVENT_CTX(cmd_list, ao_xegtao); + R_ASSERT(m_xegtao); + auto& xe = *m_xegtao; + auto* context = HW.get_context(cmd_list.context_id); + + // Match cl_pos_decompress_params and gbuffer_load_data, including jitter. + // Scale the NDC span for padding, keeping actual pixels at their original positions. + const float vertical = -tanf(deg2rad(Device.fFOV / 2.f)); + const float horizontal = -vertical / Device.fASPECT; + const bool temporal = ps_r_pp_aa_mode == TAA || ps_r_pp_aa_mode == DLSS || ps_r_pp_aa_mode == FSR3; + const Fvector4 data[3] = { + {float(xe.width), float(xe.height), 1.f / xe.width, 1.f / xe.height}, + {float(m_renderWidth), float(m_renderHeight), float(temporal ? Device.dwFrame % 64 : 0), ps_r_xegtao_radius}, + {2.f * horizontal * xe.width / m_renderWidth, 2.f * vertical * xe.height / m_renderHeight, + -horizontal * (1.f + ps_r_taa_jitter.x), -vertical * (1.f - ps_r_taa_jitter.y)}}; + D3D11_MAPPED_SUBRESOURCE mapped{}; + CHK_DX(context->Map(xe.constants.Get(), 0, D3D11_MAP_WRITE_DISCARD, 0, &mapped)); + CopyMemory(mapped.pData, data, sizeof(data)); + context->Unmap(xe.constants.Get(), 0); + + // Isolate direct compute bindings from the backend's graphics state cache. + // ClearState also removes any previous SRV/RTV aliases before binding UAVs. + context->ClearState(); + ID3D11Buffer* cb = xe.constants.Get(); + ID3D11SamplerState* sampler = xe.pointClamp.Get(); + context->CSSetConstantBuffers(0, 1, &cb); + context->CSSetSamplers(0, 1, &sampler); + ID3D11ShaderResourceView* noSRV[2]{}; + ID3D11UnorderedAccessView* noUAV[5]{}; + + { + PIX_EVENT_CTX(cmd_list, ao_xegtao_prefilter); + ID3D11ShaderResourceView* input = rt_Position->pTexture->get_SRView(); + ID3D11UnorderedAccessView* output[5] = { + xe.depth.uav[0].Get(), xe.depth.uav[1].Get(), xe.depth.uav[2].Get(), xe.depth.uav[3].Get(), xe.depth.uav[4].Get()}; + context->CSSetShader(xe.prefilter->sh, nullptr, 0); + context->CSSetShaderResources(0, 1, &input); + context->CSSetUnorderedAccessViews(0, 5, output, nullptr); + context->Dispatch(xe.width / 16, xe.height / 16, 1); + context->CSSetUnorderedAccessViews(0, 5, noUAV, nullptr); + } + { + PIX_EVENT_CTX(cmd_list, ao_xegtao_evaluate); + ID3D11ShaderResourceView* inputs[2] = {rt_Position->pTexture->get_SRView(), xe.depth.srv.Get()}; + ID3D11UnorderedAccessView* outputs[2] = {xe.ao.uav[0].Get(), xe.edges.uav[0].Get()}; + context->CSSetShader(xe.evaluate->sh, nullptr, 0); + context->CSSetShaderResources(0, 2, inputs); + context->CSSetUnorderedAccessViews(0, 2, outputs, nullptr); + context->Dispatch(xe.width / 8, xe.height / 8, 1); + context->CSSetUnorderedAccessViews(0, 2, noUAV, nullptr); + context->CSSetShaderResources(0, 2, noSRV); + } + { + PIX_EVENT_CTX(cmd_list, ao_xegtao_denoise); + ID3D11ShaderResourceView* inputs[2] = {xe.ao.srv.Get(), xe.edges.srv.Get()}; + ID3D11UnorderedAccessView* output = xe.filtered.uav[0].Get(); + context->CSSetShader(xe.denoise->sh, nullptr, 0); + context->CSSetShaderResources(0, 2, inputs); + context->CSSetUnorderedAccessViews(0, 1, &output, nullptr); + context->Dispatch(xe.width / 16, xe.height / 8, 1); + } + + context->ClearState(); + cmd_list.Invalidate(); + cmd_list.set_ColorWriteEnable(); + { + PIX_EVENT_CTX(cmd_list, ao_xegtao_export); + RenderScreenTriangle(cmd_list, rt_ao, xe.exportAO->E[0]); + } +} diff --git a/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj b/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj index d45a74e409..87fe4d5d2f 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj +++ b/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj @@ -438,6 +438,8 @@ + + @@ -484,4 +486,4 @@ - \ No newline at end of file + diff --git a/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj.filters b/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj.filters index 22597e87dc..cb4bd2ee48 100644 --- a/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj.filters +++ b/ogsr_engine/Layers/xrRenderPC_R4/xrRender_R4.vcxproj.filters @@ -849,6 +849,12 @@ Core_Target + + Core_Target + + + Core_Target + Core_Target @@ -1156,4 +1162,4 @@ imgui_render - \ No newline at end of file + diff --git a/ogsr_engine/xrGame/Weapon.cpp b/ogsr_engine/xrGame/Weapon.cpp index 66b95e96e5..2145b122c1 100644 --- a/ogsr_engine/xrGame/Weapon.cpp +++ b/ogsr_engine/xrGame/Weapon.cpp @@ -988,8 +988,9 @@ void CWeapon::UpdateDof(float& type, const Fvector4& params_type, const bool des else type += Device.fTimeDelta / dof_transition_time; - shader_exports.set_dof_params(params_type.x * type, params_type.y * type, params_type.z * type, params_type.w * type); + // Last fade-out frame must export zeros; the caller stops once type hits 0. clamp(type, 0.f, 1.f); + shader_exports.set_dof_params(params_type.x * type, params_type.y * type, params_type.z * type, params_type.w * type); } void CWeapon::UpdateLaser() diff --git a/ogsr_engine/xrGame/embedded_editor/embedded_editor_main.cpp b/ogsr_engine/xrGame/embedded_editor/embedded_editor_main.cpp index 8bb61f2d27..f3cececaaa 100644 --- a/ogsr_engine/xrGame/embedded_editor/embedded_editor_main.cpp +++ b/ogsr_engine/xrGame/embedded_editor/embedded_editor_main.cpp @@ -22,6 +22,74 @@ #include "../Layers/xrRender/Debug/dxPixEvents.h" +#include +#include + +#ifdef TRACY_ENABLE +static double pix_event_ms(const pix_event_stats& event) +{ + if (!event.freq) + return 0.0; + return 1000.0 * double(event.end - event.begin) / double(event.freq); +} + +static bool pix_event_name_matches(const xr_string& name, const char* filter) +{ + if (!filter || !*filter) + return true; + + const char* n = name.c_str(); + const size_t flen = strlen(filter); + for (size_t i = 0; n[i]; ++i) + { + size_t k = 0; + while (k < flen && n[i + k] && tolower(static_cast(n[i + k])) == tolower(static_cast(filter[k]))) + ++k; + if (k == flen) + return true; + } + return false; +} + +static void export_gpu_passes_csv(const pix_events_perf& perf, xr_string& status) +{ + xr_string csv = "index,stack,name,time_ms\n"; + for (size_t i = 0; i < perf.count; ++i) + { + const auto& event = perf.events[i]; + string512 line{}; + xr_sprintf(line, "%u,%u,\"%s\",%.4f\n", u32(i), u32(event.stack), event.name.c_str(), pix_event_ms(event)); + csv += line; + } + + ImGui::SetClipboardText(csv.c_str()); + + time_t now = time(nullptr); + tm local{}; + localtime_s(&local, &now); + + string_path rel{}; + xr_sprintf(rel, "gpu_passes_%04d%02d%02d_%02d%02d%02d.csv", local.tm_year + 1900, local.tm_mon + 1, local.tm_mday, local.tm_hour, local.tm_min, local.tm_sec); + + string_path path{}; + FS.update_path(path, fsgame::app_data_root, rel); + VerifyPath(path); + + IWriter* w = FS.w_open(path); + if (!w) + { + status = "Copied to clipboard; failed to write "; + status += path; + return; + } + + w->w(csv.c_str(), csv.size()); + FS.w_close(w); + + status = path; + status += " (copied to clipboard)"; +} +#endif CImGuiEditor::CImGuiEditor() { CImGuiGameWnd* game_wnd = xr_new(); @@ -218,21 +286,50 @@ void CImGuiMainWnd::Render() #ifdef TRACY_ENABLE ImGui::Separator(); + ImGui::TextUnformatted("GPU passes"); static int stack_levels = 8; - ImGui::SliderInt("Depth", &stack_levels, 0, 8); + static char gpu_filter[128]{}; + static xr_string gpu_export_status; + ImGui::SliderInt("Depth", &stack_levels, 0, 8); + ImGui::InputTextWithHint("##gpu_filter", "Filter by name...", gpu_filter, sizeof(gpu_filter)); + ImGui::SameLine(); auto& perf = PIXEventsStatistics(); + if (ImGui::Button("Export CSV")) + export_gpu_passes_csv(perf, gpu_export_status); + if (ImGui::IsItemHovered()) + ImGui::SetTooltip("Current frame GPU passes (all depths) to app_data and clipboard"); + + if (!gpu_export_status.empty()) + ImGui::TextWrapped("%s", gpu_export_status.c_str()); + + size_t visible = 0; for (size_t i = 0; i < perf.count; i++) { auto& event = perf.events[i]; - if (event.stack < static_cast(stack_levels)) - { - u64 time_micros = (event.end - event.begin) / (event.freq / 1000000); - float time_milliseconds = (float)time_micros * 0.001f; - ImGui::Text("%*s%s: %.3fms", event.stack * 2, " ", event.name.c_str(), time_milliseconds); - } + if (event.stack < static_cast(stack_levels) && pix_event_name_matches(event.name, gpu_filter)) + ++visible; + } + ImGui::Text("%u / %u events", u32(visible), u32(perf.count)); + + ImGui::BeginChild("gpu_pass_list", ImVec2(0, 0), ImGuiChildFlags_Borders); + + if (!visible) + ImGui::TextDisabled("No matching events"); + + for (size_t i = 0; i < perf.count; i++) + { + auto& event = perf.events[i]; + if (event.stack >= static_cast(stack_levels)) + continue; + if (!pix_event_name_matches(event.name, gpu_filter)) + continue; + + ImGui::Text("%*s%s: %.3fms", int(event.stack) * 2, " ", event.name.c_str(), pix_event_ms(event)); } + + ImGui::EndChild(); #endif RenderEnd();