#pragma once
#include "shared.h"
enum DebugViewMode {
NONE = 0,
DEPTH_grayscale = 1,
DEPTH_color = 2,
UV = 3,
NORMAL_TEX = 4,
NORMAL_GEO = 5,
TANGENT_GEO = 6,
BITANGENT_GEO = 7,
TANGENT_GEO_W = 8,
SHADING_NORMAL = 9,
ALPHA = 10,
EMISSIVE = 12,
ALBEDO = 13,
METALLIC = 14,
ROUGHNESS = 15,
OCCLUSION = 11,
SHADOW_CASCADE = 16,
LIGHTING_ONLY = 17,
CLUSTER_HEATMAP = 18,
CLUSTER_BOUNDS = 19,
WIREFRAME = 20,
PROBES_RAY_SAMPLES = 21,
PROBES_IRRADIANCE = 22,
PROBES_DISTANCE = 23,
PROBES_CLASSIFICATION = 24
};
struct Vertex {
vec3 pos;
f32 _pad0;
vec3 normal;
f32 _pad1;
vec2 uv;
vec2 _padUv; // pad 40 ->
vec4 tangent; // w handedness
};
struct CameraGPU {
mat4 view;
mat4 inv_view;
mat4 proj;
mat4 inv_proj;
mat4 view_proj;
mat4 inv_view_proj;
vec4 position_ws;
vec3 forward_ws;
f32 near_plane;
f32 far_plane;
f32 screen_w;
f32 screen_h;
};
struct ShadowCascadeGPU {
mat4 view;
mat4 view_proj;
vec3 caster_min_ls;
f32 _pad0; // 16
vec3 caster_max_ls;
f32 texel_depth;
GPU_HND(Texture2D) img;
f32 texel_size;
u32 _tail0; // 176 stride preserved (was u32 img_idx + 2 tails)
};
struct ShadowCascadesGPU {
u32 cascade_count;
// NOTE: cascade array needs absolute 16-alignment, shadows 784 in FrameUBO,
// relative offset 16 lands at 800, FIX IT!!!
u32 _pad0;
u32 _pad1;
u32 _pad2;
ShadowCascadeGPU cascade[4];
};
struct PointLightGPU {
vec3 position;
f32 range;
vec3 color;
f32 intensity;
};
struct ClusterBounds {
vec4 minPoint; // empty w
vec4 maxPoint; // empty w
};
struct ClusterRecord {
u32 lightOffset; // offset into the lights array
u32 lightCount; // total light count
};
struct RenderItemGPU {
u32 instance_id; // index into the per-frame transform buffer
u32 submesh_index;
u32 material_index;
u32 entity_id;
};
struct SubmeshGPU {
u32 first_index;
u32 index_count;
u32 base_vertex;
u32 _pad0; // 16
vec4 local_sphere; // xyz = sphere center, w = radius
vec3 local_aabb_min;
f32 _pad1;
vec3 local_aabb_max;
f32 _pad2;
};
struct TransformGPU {
mat4 world;
mat4 normal_world;
u32 active;
u32 _pad0;
u32 _pad1;
u32 _pad2;
};
struct MaterialGPU {
GPU_HND(Texture2D) albedo; // 0
GPU_HND(Texture2D) normal; // 8
GPU_HND(Texture2D) metal_rough; // 16
GPU_HND(Texture2D) emissive; // 24
u32 _padA; // 32
f32 metallic_factor; // 36
f32 roughness_factor; // 40
u32 _padB; // 44
vec3 emissive_factor; // 48
f32 _pad0; // 60
vec4 base_color; // 64..80
};
#ifndef __SLANG__
static_assert(sizeof(MaterialGPU) == 80, "MaterialGPU size drift");
static_assert(offsetof(MaterialGPU, emissive_factor) == 48, "MaterialGPU layout drift");
static_assert(offsetof(MaterialGPU, base_color) == 64, "MaterialGPU layout drift");
#endif
struct ClusterGrid {
u32 cluster_count_x;
u32 cluster_count_y;
u32 cluster_count_z;
u32 cluster_total; // total number of clusters
GPU_PTR(ClusterBounds) clusters_bounds;
u32 _tail0; // 32
u32 _tail1;
};
struct LightCull {
GPU_PTR(ClusterRecord) clusters_records;
GPU_PTR(u32) light_index_list;
};
struct DdgiGridParams {
vec4 origin; // xyz = world-space origin, w = unused
vec4 spacing; // xyz = per-axis probe spacing, w = max(spacing)
vec4 dims; // xyz = grid dimensions X/Y/Z (float for cross-lang compat), w = unused
u32 probe_count;
u32 _pad0;
u32 _pad1;
u32 _pad2;
};
struct DdgiConfig {
u32 enable_accumulation = 1;
f32 history_alpha = 0.97f;
f32 irradiance_encoding_gamma = 5.0f;
f32 irradiance_threshold = 0.25f;
f32 brightness_threshold = 2.0f;
f32 probe_min_frontface_distance = 0.1f; // world meters
f32 probe_distance_scale = 1.0f; // world meters
f32 fixed_ray_backface_threshold = 0.25f;
f32 random_ray_backface_threshold = 0.25f;
u32 _tail0 = 0; // round struct size to 16 (36 -> 48, std140 nested struct)
u32 _tail1 = 0;
u32 _tail2 = 0;
};
struct FrameUBO {
// camera
mat4 view;
mat4 inv_view;
mat4 proj;
mat4 inv_proj;
mat4 view_proj;
mat4 inv_view_proj;
vec4 position_ws;
vec3 forward_ws;
f32 near_plane;
f32 far_plane;
f32 screen_w;
f32 screen_h;
u32 _pad0;
// meshes
GPU_PTR(RenderItemGPU) render_items;
GPU_PTR(SubmeshGPU) submeshes;
GPU_PTR(MaterialGPU) materials;
GPU_PTR(TransformGPU) transforms;
// env (handles, rhi::handle_id filled)
GPU_HND(TextureCube<float4>) env;
GPU_HND(Texture2D) env_brdf;
GPU_HND(TextureCube<float4>) env_irradiance;
GPU_HND(TextureCube<float4>) env_prefiltered;
u32 env_prefilter_mip;
u32 _pad1;
u32 _pad2;
u32 _pad3;
mat4 dbg_view_proj;
mat4 dbg_inv_view_proj;
vec3 dbg_pos_ws;
f32 _pad4;
vec3 dbg_fwd_ws;
f32 _pad5;
vec3 light_dir;
f32 _pad6;
vec3 light_color;
f32 light_intensity;
mat4 main_light_view_proj;
GPU_PTR(PointLightGPU) point_lights;
u32 point_light_count;
u32 _pad_pl0;
u32 _pad_pl1;
u32 _pad_pl2;
u32 _pad_pl3;
u32 _pad_pl4;
ShadowCascadesGPU shadows;
ClusterGrid cluster;
LightCull light_cull;
DdgiGridParams ddgi_grid;
DdgiConfig ddgi_cfg;
u64 tlas_address = 0;
DebugViewMode debug = DebugViewMode::NONE;
GPU_HND(SamplerState) aniso_wrap_mips; // 24..32 (slot 1)
GPU_HND(SamplerState) linear_clamp_mips; // 32..40 (slot 2)
GPU_HND(SamplerState) linear_clamp; // 40..48 (slot 3)
GPU_HND(SamplerComparisonState) cmp_greater; // 48..56 (slot 0, comparison view for shadows)
};
#ifndef __SLANG__
static_assert(sizeof(FrameUBO) % 16 == 0, "FrameUBO size must stay 16-multiple");
#endif
struct InstanceData {
u32 instance_id;
u32 render_item_idx;
};
// One visible (post-culling) draw entry produced by build_indirect.
struct VisibleItem {
u32 id; // render-item index
u32 instance_id; // instance/transform index
};
// Matches VkDrawIndexedIndirectCommand (5 x u32 = 20 bytes).
struct DrawIndexedIndirectCommand {
u32 indexCount;
u32 instanceCount;
u32 firstIndex;
u32 vertexOffset;
u32 firstInstance;
};
struct ProbeState {
vec4 offset_active; // xyz = location offset, w = active 0/1
};
struct FwdPC {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
};
struct FwdDrawPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_HND(StructuredBuffer<Vertex>) vertex;
GPU_PTR(VisibleItem) visible_items;
GPU_PTR(ProbeState) ddgi_probe_state;
GPU_HND(Texture2D) ddgi_irradiance;
GPU_HND(Texture2D) ddgi_distance;
};
struct IdBlitPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_HND(StructuredBuffer<Vertex>) vertex;
GPU_PTR(VisibleItem) visible_items;
GPU_HND(Texture2D) depth;
};
struct EquirectPushConstants {
u32 faceidx;
u32 hdrTextureIdx;
};
struct IrradiancePushConstants {
mat4 viewProj;
u32 faceidx;
u32 environmentIdx;
};
struct PrefilterPushConstants {
mat4 viewProj;
u32 faceidx;
u32 environmentIdx;
f32 roughness;
f32 environment_resolution;
};
struct IblConstants {
mat4 viewProj;
f32 roughness;
f32 environment_resolution;
u32 faceidx;
GPU_HND(Texture2D) hdrTexture;
GPU_HND(TextureCube<float4>) environment;
u32 _pad;
GPU_HND(SamplerState) samp; // rhi::handle_id (init-time passes have no frame access)
};
struct FrustumDebugPushConstants {
mat4 debug_view_proj;
mat4 debug_inv_view_proj;
u32 frame_idx;
u32 _pad0;
};
struct BuildIndirectPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_PTR(DrawIndexedIndirectCommand) indirect_buffer;
GPU_PTR(VisibleItem) visible_items;
GPU_PTR(u32) draw_count_buffer;
u32 item_count;
GPU_HND(Texture2D) hiz; // UINT64_MAX disables (mint failure / culling off)
u32 hiz_width;
u32 hiz_height;
u32 hiz_mip_count;
u32 enable_culling;
};
struct FullscreenPushConstants {
GPU_HND(Texture2D) lighting;
GPU_HND(Texture2D) depth;
GPU_HND(Texture2D) ssr;
GPU_HND(SamplerState) samp; // rhi::handle_id, uniform (no frame access in this pass)
};
struct SkyboxPushConstants {
GPU_HND(TextureCube<float4>) cubemap;
u32 depth_tex_idx;
GPU_HND(ConstantBuffer<FrameUBO>) frame;
};
struct GridPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
};
struct DebugLineVertexGPU {
vec3 pos;
u32 packed; // [category:8][depth:8][reserved:16]
vec4 color;
u32 width_bits; // f32 screen-space width in px (thick quad expansion)
u32 _pad1; // reserved: gizmo id
};
struct DebugLinesPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_PTR(DebugLineVertexGPU) lines;
f32 viewport_w;
f32 viewport_h;
};
struct ImGuiPushConstants {
vec2 scale; // FramebufferScale (DisplaySize * scale = fb size)
vec2 translate; // -DisplayPos * scale
vec2 display_size; // framebuffer extent
GPU_HND(Texture2D) tex; // bindless handle of font/user texture
GPU_HND(SamplerState) samp; // rhi::handle_id, uniform (no frame access in this pass)
};
// ---- cluster ----
STATIC_CONST u32 MAX_POINT_LIGHTS = 2048;
STATIC_CONST u32 NUM_POINT_LIGHTS = 512;
// max lights per cluster
// have to make sure this has enough of an additional buffer break down the math since all of this nonsense is
// duplicated for each FIF
STATIC_CONST u32 MAX_LIGHT_INDICES = MAX_POINT_LIGHTS * MAX_POINT_LIGHTS;
// cluster grid constants
STATIC_CONST u32 TILE_SIZE_X = 64;
STATIC_CONST u32 TILE_SIZE_Y = 64;
STATIC_CONST u32 CLUSTER_COUNT_Z = 32;
struct ClusterBoundsPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
};
struct LightCullPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
u32 light_count;
u32 _pad0; // keep BDA 8-aligned
GPU_PTR(u32) light_index_counter; // single uint counter
u32 phase; // 0 = count, 1 = fill
u32 _tail; // round size to 8 (struct holds u64s: 28 -> 32)
};
// ---- shadows ----
struct ShadowPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_PTR(VisibleItem) visible_items;
u32 cascade_idx;
GPU_HND(StructuredBuffer<Vertex>) vertex;
};
// ---- ssr ----
struct SsrPushConstants {
GPU_HND(Texture2D) color;
GPU_HND(Texture2D) depth;
GPU_HND(Texture2D) normal;
GPU_HND(RWTexture2D<float4>) output;
GPU_HND(ConstantBuffer<FrameUBO>) frame;
u32 width;
u32 height;
u32 _pad0;
f32 max_dist; // march distance cap, world units
f32 thickness; // hit epsilon in post-projection depth units (reverse-Z)
i32 max_steps; // fixed march iteration count
f32 stride; // world-space step per iteration
u32 _tail; // round size to 8 (struct holds u64s)
};
#ifndef __SLANG__
static_assert(offsetof(SsrPushConstants, frame) == 32);
static_assert(offsetof(SsrPushConstants, max_dist) == 52);
static_assert(sizeof(SsrPushConstants) == 72);
static_assert(sizeof(SsrPushConstants) <= 128);
#endif
// ---- hiz ----
struct HiZPushConstants {
GPU_HND(Texture2D) depth; // bindless handle of depth texture (for level 1 source)
GPU_HND(Texture2D) hiz_src; // bindless handle of hi-z texture (sampled) for higher level source
GPU_HND(RWTexture2D<float4>) hiz_dst; // bindless handle of current mip's storage image (writable)
u32 mip_level; // 1 = first level to generate
u32 src_width; // width at current mip level's source
u32 src_height; // height at current mip level's source
u32 _tail; // round size to 8 (struct holds u64s: 36 -> 40)
};
// ---- ddgi ----
STATIC_CONST u32 RAYS_PER_PROBE = 128;
STATIC_CONST u32 FIXED_RAY_COUNT = 32;
STATIC_CONST u32 ATLAS_SIZE = 16;
STATIC_CONST u32 IRRADIANCE_SIZE = 8;
STATIC_CONST int DDGI_TILE_BORDER = 2; // 2 pixel border around the tile for better alignment
STATIC_CONST int IRRADIANCE_TILE_STRIDE = IRRADIANCE_SIZE + DDGI_TILE_BORDER;
STATIC_CONST int DISTANCE_TILE_STRIDE = ATLAS_SIZE + DDGI_TILE_BORDER;
STATIC_CONST f32 HISTORY_ALPHA = 0.97;
// 1 = inactive, 0 = active (inverted)
STATIC_CONST f32 DDGI_STATE_PROBE_INACTIVE = 1.0f;
STATIC_CONST f32 DDGI_STATE_PROBE_ACTIVE = 0.0f;
struct ProbeRaySample {
vec4 radiance_distance; // rgb = radiance, a = hit distance
vec4 direction; // xyz = ray direction, a = unused
};
struct DdgiAtlasMeta {
u32 tiles_per_row;
u32 row_count;
u32 width;
u32 height;
};
struct ProbeSpherePushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_PTR(vec4) sphere_mesh;
u32 sphere_vertex_count;
GPU_PTR(ProbeRaySample) samples_buffer;
GPU_HND(Texture2D) atlas_sampled;
GPU_PTR(ProbeState) probe_state;
u32 rays_per_probe;
DebugViewMode debug_flag;
};
struct DdgiTracePushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
u64 framecount_idx;
u32 tlas_idx;
GPU_PTR(ProbeRaySample) samples_buffer;
GPU_PTR(InstanceData) instance_data;
GPU_HND(Texture2D) irradiance;
GPU_HND(Texture2D) distance;
GPU_PTR(ProbeState) probe_state;
u32 cascade_count;
GPU_HND(StructuredBuffer<Vertex>) vertex;
GPU_PTR(u32) index_bda;
u32 _pad0; // align random_rotation to 16 (std430 vec4 after u64 block ends @72)
u32 _pad1;
vec4 random_rotation; // xyz = random axis, w = random angle [0, 2PI]
};
struct DdgiUpdatePushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
u32 frame_counter;
GPU_PTR(ProbeRaySample) samples_buffer;
GPU_PTR(ProbeState) probe_state;
GPU_HND(Texture2D) atlas_sample;
GPU_HND(RWTexture2D<float4>) atlas_storage;
};
struct DdgiRelocatePushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
GPU_PTR(ProbeRaySample) samples_buffer;
GPU_PTR(ProbeState) probe_state;
};
struct DdgiClassifyPushConstants {
GPU_HND(ConstantBuffer<FrameUBO>) frame;
u64 frame_counter;
GPU_PTR(ProbeRaySample) samples_buffer;
GPU_PTR(ProbeState) probe_state;
};
#ifndef __SLANG__
static_assert(offsetof(DdgiTracePushConstants, frame) == 0);
static_assert(offsetof(DdgiTracePushConstants, framecount_idx) == 8);
static_assert(offsetof(DdgiTracePushConstants, vertex) == 72);
static_assert(offsetof(DdgiTracePushConstants, index_bda) == 80);
static_assert(offsetof(DdgiTracePushConstants, random_rotation) == 96);
static_assert(sizeof(DdgiTracePushConstants) == 112);
#endif
#ifdef __SLANG__
struct VtxOut {
vec4 clip_pos : SV_Position;
vec4 view_pos : TEXCOORD0;
vec3 world_pos : TEXCOORD1;
vec3 normal : TEXCOORD2;
vec4 tangent : TEXCOORD3;
vec2 uv : TEXCOORD4;
nointerpolation uint material_index : MATERIAL_INDEX;
vec4 clip_debug : TEXCOORD7;
// nointerpolation f32 frustum_pass : TEXCOORD9;
// nointerpolation f32 visible : TEXCOORD10;
};
#endif