Pseudo random background (#51)

* working pseudorandom

* Fix comment length in SsVulkan.cmake

* Fix comment lenght
This commit is contained in:
MotivaCG
2026-09-04 13:56:47 -04:00
committed by GitHub
parent cbb0ecf5cf
commit 5d214af9f5
14 changed files with 285 additions and 85 deletions
-2
View File
@@ -83,8 +83,6 @@ function(_ss_vulkan_header_version inc out_var)
endif()
endfunction()
# _ss_fetch_vulkan_headers(<include_var>)
#
# Unpacks the pinned Vulkan-Headers release into the build tree. Header-only
# and architecture-independent, so this covers every platform whose loader is
# current but whose headers are not.
+4 -1
View File
@@ -489,7 +489,7 @@ EngineStepConfig build_step_config(const TrainConfig& c, const RunState& st, int
cfg.ppisp.run_before_color_space = c.apply_ppisp_before_color_space;
// ---- background ----------------------------------------------------
if (c.background_mode == "noise") {
if (c.background_mode == "noise" || c.background_mode == "pseudorandom") {
float rw = std::min((float)step / std::max(c.background_noise_warmup, 1), 1.0f);
cfg.background.randomize_weight =
1.0f - (1.0f - c.background_noise_pre_warmup) * (1.0f - rw);
@@ -795,6 +795,9 @@ void TrainerSession::setup_engine() {
if (cfg.background_mode == "noise")
engine_init_background_noise((int)color.splat_transfer,
color.splat_linear);
else if (cfg.background_mode == "pseudorandom")
engine_init_background_pseudorandom((int)color.splat_transfer,
color.splat_linear);
else if (cfg.background_mode == "sh")
engine_init_background_sh(cfg.background_sh_degree,
(int)color.splat_transfer,
+4 -2
View File
@@ -97,12 +97,14 @@ int main(int argc, char** argv) {
readback_f(acc, d_vb, PIX * 3);
}
// ---- blend_background_noise_backward (every mode x reg off/on) ----
// ---- blend_background_noise_backward (every mode x draw x reg off/on) ----
for (int xf = 0; xf < 5; xf++) for (int lin = 0; lin < 2; lin++)
for (int blocky = 0; blocky < 2; blocky++)
for (float over : {0.0f, 3.0f}) {
float* d_vr = fresh3();
float* d_vt = fresh1();
blend_background_noise_backward(xf, lin != 0, t3(d_rgb), t1(d_T), 0.7f,
blend_background_noise_backward(xf, lin != 0, blocky != 0,
t3(d_rgb), t1(d_T), 0.7f,
1234u + xf, over, t3(d_vout),
t3(d_vr), t1(d_vt));
backend::device_synchronize();
@@ -22,9 +22,9 @@ static_assert(sizeof(BlendBgParams) == 4 * 8 + 2 * 4, "layout");
struct BlendBgNoiseParams {
uint64_t rgb, transmittance, out_rgb;
float randomize_weight;
uint32_t seed, HW, total, wgs_per_row;
uint32_t seed, HW, total, wgs_per_row, W, blocky;
};
static_assert(sizeof(BlendBgNoiseParams) == 3 * 8 + 5 * 4 + 4 /*pad*/,
static_assert(sizeof(BlendBgNoiseParams) == 3 * 8 + 7 * 4 + 4 /*pad*/,
"layout");
// Mirrors RgbToSrgbParams.
@@ -68,6 +68,7 @@ void blend_background_forward(
void blend_background_noise_forward(
int transfer,
bool is_linear,
bool blocky,
DeviceTensor3D<float3> rgb,
DeviceTensor3D<float> transmittance,
float randomize_weight,
@@ -84,6 +85,8 @@ void blend_background_noise_forward(
p.seed = seed;
p.HW = (uint32_t)hw;
p.total = (uint32_t)total;
p.W = (uint32_t)rgb.size<2>();
p.blocky = blocky ? 1u : 0u;
vkk::dispatch_flat("pixel_wise_render.blend_background_noise_fwd",
backend::vk::SpecList{(uint32_t)transfer,
is_linear ? 1u : 0u},
@@ -26,9 +26,11 @@ static_assert(sizeof(BlendBgBwdParams) == 7 * 8 + 3 * 4 + 4 /*pad*/,
struct BlendBgNoiseBwdParams {
uint64_t rgb, transmittance, v_out_rgb, v_rgb, v_transmittance;
float overexposure_scale, randomize_weight;
uint32_t seed, HW, total, wgs_per_row;
uint32_t seed, HW, total, wgs_per_row, W, blocky;
};
static_assert(sizeof(BlendBgNoiseBwdParams) == 5 * 8 + 6 * 4, "layout");
// 40 + 32 lands on an 8 boundary, so unlike the branch's 7-field version this
// one needs no tail padding.
static_assert(sizeof(BlendBgNoiseBwdParams) == 5 * 8 + 8 * 4, "layout");
// Mirrors RgbToSrgbBwdParams.
struct RgbToSrgbBwdParams {
@@ -123,6 +125,7 @@ void blend_background_backward(
void blend_background_noise_backward(
int transfer,
bool is_linear,
bool blocky,
DeviceTensor3D<float3> rgb,
DeviceTensor3D<float> transmittance,
float randomize_weight,
@@ -146,6 +149,8 @@ void blend_background_noise_backward(
p.seed = seed;
p.HW = (uint32_t)hw;
p.total = (uint32_t)total;
p.W = (uint32_t)rgb.size<2>();
p.blocky = blocky ? 1u : 0u;
vkk::dispatch_flat("pixel_wise_train.blend_bg_noise_bwd",
backend::vk::SpecList{(uint32_t)transfer,
is_linear ? 1u : 0u},
@@ -76,7 +76,45 @@ uint hash_uint3(uint a, uint b, uint c) {
return hash;
}
struct BlendBgNoiseParams { // 44 bytes: pushed directly
// Side, in pixels, of a `blocky` background tile. Per-pixel noise is averaged
// back into flat grey by SSIM's 11x11 window and by the multi-scale pyramid,
// which is where the penalty is supposed to land; a tile survives both.
static const uint kBgBlockPx = 64u;
// Murmur-style finalizer. The whole background is computed, never stored.
[ForceInline]
uint _bg_mix(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// The unit sample the background is built from. `blocky` draws one of the 8 RGB
// cube corners per tile -- the extreme points of the cube, so residual
// transparency costs the most it can. Twin of kernels/pixelwise/ImageColorOps.cu.
[ForceInline]
float3 _bg_sample(uint blocky, uint seed, uint gid, uint bid, uint W) {
if (blocky != 0u) {
// The tile grid shifts every step, so no pixel keeps its colour and
// the pattern cannot be baked into the splats.
uint ox = _bg_mix(seed * 2u + 1u) % kBgBlockPx;
uint oy = _bg_mix(seed * 2u + 7u) % kBgBlockPx;
uint h = _bg_mix((((gid % W) + ox) / kBgBlockPx) * 2654435761u
^ (((gid / W) + oy) / kBgBlockPx) * 40503u
^ (seed + bid * 0x9e3779b9u));
return float3((h & 1u) != 0u ? 1.0f : -1.0f,
(h & 2u) != 0u ? 1.0f : -1.0f,
(h & 4u) != 0u ? 1.0f : -1.0f);
}
// 2u-1: without it the plain path lands in [0.5, 0.5+w/2) instead of
// straddling 0.5, so every channel sits in the same bright half.
return float3(float(hash_uint3(seed + 0, gid, bid)) * exp2(-31.0f) - 1.0f,
float(hash_uint3(seed + 1, gid, bid)) * exp2(-31.0f) - 1.0f,
float(hash_uint3(seed + 2, gid, bid)) * exp2(-31.0f) - 1.0f);
}
struct BlendBgNoiseParams { // 56 bytes: pushed directly
float* rgb; // [B,H,W,3]
float* transmittance; // [B,H,W,1]
float* out_rgb; // [B,H,W,3]
@@ -85,6 +123,8 @@ struct BlendBgNoiseParams { // 44 bytes: pushed directly
uint32_t HW; // H*W (hash uses per-image pixel id + batch id)
uint32_t total; // B*H*W
uint32_t wgs_per_row;
uint32_t W;
uint32_t blocky;
};
[shader("compute")]
@@ -101,11 +141,8 @@ void blend_background_noise_fwd(uint3 wg: SV_GroupID,
p.rgb[3 * idx + 2]);
float T = p.transmittance[idx];
float3 background;
background.x = float(hash_uint3(p.seed + 0, gid, bid)) * exp2(-32.0f);
background.y = float(hash_uint3(p.seed + 1, gid, bid)) * exp2(-32.0f);
background.z = float(hash_uint3(p.seed + 2, gid, bid)) * exp2(-32.0f);
background = 0.5f + 0.5f * p.randomize_weight * (2.0f * background - 1.0f);
float3 background = _bg_sample(p.blocky, p.seed, gid, bid, p.W);
background = 0.5f + 0.5f * p.randomize_weight * background;
background = display_to_working3(background, kTransfer, kIsLinear != 0);
rgb = blend_background(rgb, T, background);
@@ -93,7 +93,45 @@ uint _pwt_hash_uint3(uint a, uint b, uint c) {
return hash;
}
struct BlendBgNoiseBwdParams { // 64 bytes: pushed directly
// Side, in pixels, of a `blocky` background tile. Per-pixel noise is averaged
// back into flat grey by SSIM's 11x11 window and by the multi-scale pyramid,
// which is where the penalty is supposed to land; a tile survives both.
static const uint kBgBlockPx = 64u;
// Murmur-style finalizer. The whole background is computed, never stored.
[ForceInline]
uint _bg_mix(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// The unit sample the background is built from. `blocky` draws one of the 8 RGB
// cube corners per tile -- the extreme points of the cube, so residual
// transparency costs the most it can. Twin of kernels/pixelwise/ImageColorOps.cu.
[ForceInline]
float3 _bg_sample(uint blocky, uint seed, uint gid, uint bid, uint W) {
if (blocky != 0u) {
// The tile grid shifts every step, so no pixel keeps its colour and
// the pattern cannot be baked into the splats.
uint ox = _bg_mix(seed * 2u + 1u) % kBgBlockPx;
uint oy = _bg_mix(seed * 2u + 7u) % kBgBlockPx;
uint h = _bg_mix((((gid % W) + ox) / kBgBlockPx) * 2654435761u
^ (((gid / W) + oy) / kBgBlockPx) * 40503u
^ (seed + bid * 0x9e3779b9u));
return float3((h & 1u) != 0u ? 1.0f : -1.0f,
(h & 2u) != 0u ? 1.0f : -1.0f,
(h & 4u) != 0u ? 1.0f : -1.0f);
}
// 2u-1: without it the plain path lands in [0.5, 0.5+w/2) instead of
// straddling 0.5, so every channel sits in the same bright half.
return float3(float(_pwt_hash_uint3(seed + 0, gid, bid)) * exp2(-31.0f) - 1.0f,
float(_pwt_hash_uint3(seed + 1, gid, bid)) * exp2(-31.0f) - 1.0f,
float(_pwt_hash_uint3(seed + 2, gid, bid)) * exp2(-31.0f) - 1.0f);
}
struct BlendBgNoiseBwdParams { // 72 bytes: pushed directly
float* rgb; // PRE-blend
float* transmittance;
float* v_out_rgb;
@@ -105,6 +143,8 @@ struct BlendBgNoiseBwdParams { // 64 bytes: pushed directly
uint32_t HW;
uint32_t total;
uint32_t wgs_per_row;
uint32_t W;
uint32_t blocky;
};
[shader("compute")]
@@ -120,11 +160,8 @@ void blend_bg_noise_bwd(uint3 wg: SV_GroupID, uint tid: SV_GroupThreadID,
p.rgb[3 * idx + 2]);
float T = p.transmittance[idx];
float3 background;
background.x = float(_pwt_hash_uint3(p.seed + 0, gid, bid)) * exp2(-32.0f);
background.y = float(_pwt_hash_uint3(p.seed + 1, gid, bid)) * exp2(-32.0f);
background.z = float(_pwt_hash_uint3(p.seed + 2, gid, bid)) * exp2(-32.0f);
background = 0.5f + 0.5f * p.randomize_weight * (2.0f * background - 1.0f);
float3 background = _bg_sample(p.blocky, p.seed, gid, bid, p.W);
background = 0.5f + 0.5f * p.randomize_weight * background;
background = display_to_working3(background, kTransfer, kIsLinear != 0);
float3 v_out = float3(p.v_out_rgb[3 * idx], p.v_out_rgb[3 * idx + 1],
+1 -1
View File
@@ -145,7 +145,7 @@ inline int train_tier_rank(const char* tier) {
X(std::string, primitive, "3dgs", "splats", "basic", "3dgs|mip|3dgut") \
X(int, sh_degree, 3, "splats", "basic", "") \
X(int, sh_degree_warmup_every, 1000, "splats", "expert", "") \
X(std::string, background_mode, "black", "splats", "basic", "black|noise|sh") \
X(std::string, background_mode, "black", "splats", "basic", "black|noise|sh|pseudorandom") \
X(int, background_sh_degree, 4, "splats", "basic", "") \
X(int, background_noise_warmup, 2000, "splats", "expert", "") \
X(float, background_noise_pre_warmup, 0.25f, "splats", "expert", "") \
+4
View File
@@ -253,6 +253,10 @@ void engine_ppisp_optim_step(int step, const PpispStepConfig& cfg);
//
// dc_color is the linear-space DC color used at SH init time (set slot 0).
void engine_init_background_noise(int splat_transfer, bool splat_is_linear);
// Same blend, but the draw is one of the 8 RGB cube corners per tile instead
// of per-pixel noise: the extremes cost residual transparency the most, and a
// tile survives the SSIM window that averages per-pixel noise back to grey.
void engine_init_background_pseudorandom(int splat_transfer, bool splat_is_linear);
void engine_init_background_sh(
int sh_degree, int splat_transfer, bool splat_is_linear);
+17 -2
View File
@@ -33,6 +33,14 @@ void engine_init_background_noise(int splat_transfer, bool splat_is_linear) {
bg.splat_is_linear = splat_is_linear;
}
void engine_init_background_pseudorandom(int splat_transfer, bool splat_is_linear) {
auto& bg = engine().background;
bg.mode = EngineBackground::Mode::Pseudorandom;
bg.enabled = true;
bg.splat_transfer = splat_transfer;
bg.splat_is_linear = splat_is_linear;
}
// Allocates the SH parameter table; slot 0 is the DC colour.
void engine_init_background_sh(int sh_degree, int splat_transfer,
bool splat_is_linear) {
@@ -148,9 +156,11 @@ void _engine_background_forward() {
DeviceTensor3D<float3> post_rgb;
post_rgb.resize(PoolSlot::EngBgSkyRgbPost, C_batch, H, W);
if (bg.mode == EngineBackground::Mode::Noise) {
if (bg.mode == EngineBackground::Mode::Noise ||
bg.mode == EngineBackground::Mode::Pseudorandom) {
blend_background_noise_forward(
bg.splat_transfer, bg.splat_is_linear,
bg.mode == EngineBackground::Mode::Pseudorandom,
bg.fwd_pre_blend_rgb, Ts_in,
bg.cur_randomize_weight, bg.cur_seed,
post_rgb);
@@ -212,11 +222,16 @@ void _engine_background_backward_hook(
DeviceTensor3D<float3> v_rgb(v_render_rgb);
DeviceTensor3D<float> v_Ts_scratch_dt(v_Ts_scratch_tv);
if (bg.mode == EngineBackground::Mode::Noise) {
if (bg.mode == EngineBackground::Mode::Noise ||
bg.mode == EngineBackground::Mode::Pseudorandom) {
// Pre-blend, not post: the overexposure term needs the unclamped
// composite, which the blend output cannot recover. Passing post-blend
// was harmless only while the affine derivative was the sole reader.
DeviceTensor3D<float> Ts_in(fwd_Ts_tensor);
DeviceTensor3D<float3> v_out(v_render_rgb);
blend_background_noise_backward(
bg.splat_transfer, bg.splat_is_linear,
bg.mode == EngineBackground::Mode::Pseudorandom,
bg.fwd_pre_blend_rgb, Ts_in,
bg.cur_randomize_weight, bg.cur_seed,
overexposure_reg_weight,
+4 -4
View File
@@ -368,11 +368,11 @@ struct BilagridNormal {
bool quantize_value() const { return value_bits != 32; }
};
// Background blending. Applied BEFORE bilagrid/PPISP. Two modes:
// - Noise: random per-pixel color (warmup-weighted). No persistent state.
// - Sh: skybox = SH(world ray dir) + DC color. DC + L1+ coeffs trained.
// Background blending. Applied BEFORE bilagrid/PPISP. Noise and Pseudorandom
// are the same stateless blend over a different draw; Sh is a trained skybox,
// SH(world ray dir) + DC color, and carries the only persistent state here.
struct EngineBackground {
enum class Mode { None = 0, Noise = 1, Sh = 2 };
enum class Mode { None = 0, Noise = 1, Sh = 2, Pseudorandom = 3 };
Mode mode = Mode::None;
bool enabled = false;
+91 -43
View File
@@ -2351,51 +2351,99 @@ SS_MSG(background_mode,
KO("배경"), DE("Hintergrund"), FR("Arrière-plan"), ES("Fondo"),
PT("Fundo"), IT("Sfondo"), NL("Achtergrond"), RU("Фон"), TR("Arka plan"));
SS_MSG(background_mode_help,
EN("What fills pixels no splat covers. `black` is the usual choice, "
"`noise` discourages background transparency, and `sh` learns a skybox "
"so distant background is represented instead of ignored."),
JA("スプラットが覆っていない画素を何で埋めるかです。`black` が通常の選択で"
"す。`noise` は背景が透けるのを抑えます。`sh` はスカイボックスを学習し、"
"遠景を無視せずに表現します。"),
ZH_HANS("没有泼溅覆盖的像素用什么填充。`black` 是常规选择;`noise` 可以抑"
"制背景透明;`sh` 会学习一个天空盒,让远景被表示出来而不是被忽略。"),
ZH_HANT("沒有潑濺覆蓋的像素用什麼填滿。`black` 是常規選擇;`noise` 可以抑"
"制背景透明;`sh` 會學習一個天空盒,讓遠景被表示出來而不是被忽略。"),
KO("스플랫이 덮지 않은 픽셀을 무엇으로 채울지입니다. `black`이 보통 선택이"
"고, `noise`는 배경이 비치는 것을 억제하며, `sh`는 스카이박스를 학습해 "
"먼 배경을 무시하지 않고 표현합니다."),
DE("Womit Pixel gefüllt werden, die kein Splat bedeckt. `black` ist die "
"übliche Wahl, `noise` hält den Hintergrund davon ab, durchsichtig zu "
"werden, und `sh` lernt eine Skybox, sodass ferner Hintergrund "
"dargestellt statt ignoriert wird."),
EN("What fills pixels no splat covers. `black` is the usual choice. "
"`noise` and `pseudorandom` both discourage half-transparent "
"surfaces by making a pixel left uncovered land on a colour that "
"changes every step; pseudorandom draws vivid tiles rather than "
"per-pixel speckle, which the loss cannot average away, so it "
"presses harder. `sh` learns a skybox so distant background is "
"represented instead of ignored."),
JA("スプラットが覆わない画素を何で埋めるかです。`black"
"` が通常の選択です。`noise` と `"
"pseudorandom` はどちらも、覆われていない画"
"素の色が毎ステップ変わるようにして半透明な面を抑えます。"
"`pseudorandom` は画素ごとの細かいノイズで"
"はなく鮮やかなタイルを使うため、損失に平均化されず効き目"
"が強くなります。`sh` は空を学習し、遠くの背景を無視"
"せず表現します。"),
ZH_HANS("用什么填充没有泼溅覆盖的像素。`black` 是通常的选"
"择。`noise` 和 `pseudorandom` 都"
"通过让未覆盖像素的颜色每步都变来抑制半透明表面;`"
"pseudorandom` 用的是鲜艳的色块而不是逐像素"
"的细噪点,损失无法把它平均掉,所以压得更狠。`sh` 会"
"学习一个天空盒,让远处背景被表示而不是被忽略。"),
ZH_HANT("用什麼填充沒有潑濺覆蓋的像素。`black` 是通常的選"
"擇。`noise` 和 `pseudorandom` 都"
"透過讓未覆蓋像素的顏色每步都變來抑制半透明表面;`"
"pseudorandom` 用的是鮮豔的色塊而不是逐像素"
"的細雜訊,損失無法把它平均掉,所以壓得更狠。`sh` 會"
"學習一個天空盒,讓遠處背景被表示而不是被忽略。"),
KO("스플랫이 덮지 않은 픽셀을 무엇으로 채울지입니다. "
"`black`이 보통의 선택입니다. `noise`와"
" `pseudorandom`은 덮이지 않은 픽셀의 "
"색이 매 스텝 바뀌게 해서 반투명한 표면을 억제합니"
"다. `pseudorandom`은 픽셀 단위의 잔 "
"노이즈 대신 선명한 타일을 쓰므로 손실이 평균으로 "
"지워 버리지 못해 더 세게 누릅니다. `sh`는 스"
"카이박스를 학습해 먼 배경을 무시하지 않고 표현합니"
"다."),
DE("Was Pixel füllt, die kein Splat bedeckt. `black` ist die übliche "
"Wahl. `noise` und `pseudorandom` entmutigen beide "
"halbdurchsichtige Flächen, indem ein unbedecktes Pixel auf einer "
"Farbe landet, die sich jeden Schritt ändert; pseudorandom zeichnet "
"kräftige Kacheln statt Sprenkel pro Pixel, die der Verlust nicht "
"wegmitteln kann, und drückt deshalb stärker. `sh` lernt eine "
"Skybox, damit ferner Hintergrund dargestellt statt ignoriert wird."),
FR("Ce qui remplit les pixels qu'aucun splat ne couvre. `black` est le "
"choix habituel, `noise` décourage la transparence de l'arrière-plan, "
"et `sh` apprend un skybox pour que l'arrière-plan lointain soit "
"représenté au lieu d'être ignoré."),
ES("Con qué se rellenan los píxeles que ningún splat cubre. `black` es la "
"opción habitual, `noise` desalienta la transparencia del fondo, y "
"`sh` aprende un skybox para que el fondo lejano se represente en vez "
"de ignorarse."),
PT("Com o que são preenchidos os pixels que nenhum splat cobre. `black` é "
"a escolha habitual, `noise` desencoraja a transparência do fundo, e "
"`sh` aprende um skybox para que o fundo distante seja representado em "
"vez de ignorado."),
IT("Con che cosa vengono riempiti i pixel che nessuno splat copre. "
"`black` è la scelta abituale, `noise` scoraggia la trasparenza dello "
"sfondo, e `sh` impara uno skybox così lo sfondo lontano viene "
"rappresentato invece che ignorato."),
NL("Waarmee pixels worden gevuld die geen enkele splat bedekt. `black` is "
"de gebruikelijke keuze, `noise` ontmoedigt doorzichtigheid van de "
"achtergrond, en `sh` leert een skybox zodat verre achtergrond wordt "
"weergegeven in plaats van genegeerd."),
RU("Чем заполняются пиксели, не покрытые ни одним сплатом. `black` — "
"обычный выбор, `noise` не даёт фону становиться прозрачным, а `sh` "
"обучает скайбокс, чтобы дальний фон был представлен, а не "
"проигнорирован."),
"choix habituel. `noise` et `pseudorandom` découragent tous deux "
"les surfaces à demi transparentes en faisant tomber un pixel non "
"couvert sur une couleur qui change à chaque étape ; pseudorandom "
"dessine des tuiles vives plutôt qu'un grain par pixel, que la "
"perte ne peut pas moyenner, et appuie donc plus fort. `sh` apprend "
"une skybox pour que l'arrière-plan lointain soit représenté au "
"lieu d'être ignoré."),
ES("Qué rellena los píxeles que ningún splat cubre. `black` es la "
"elección habitual. `noise` y `pseudorandom` desincentivan las "
"superficies semitransparentes haciendo que un píxel sin cubrir "
"caiga sobre un color que cambia en cada paso; pseudorandom dibuja "
"baldosas vivas en vez de grano por píxel, que la pérdida no puede "
"promediar, así que aprieta más. `sh` aprende un cielo para que el "
"fondo lejano quede representado en vez de ignorado."),
PT("O que preenche os pixels que nenhum splat cobre. `black` é a "
"escolha habitual. `noise` e `pseudorandom` desencorajam "
"superfícies semitransparentes fazendo um pixel descoberto cair "
"sobre uma cor que muda a cada passo; pseudorandom desenha "
"ladrilhos vivos em vez de grão por pixel, que a perda não consegue "
"mediar, então aperta mais. `sh` aprende um céu para que o fundo "
"distante seja representado em vez de ignorado."),
IT("Che cosa riempie i pixel che nessuno splat copre. `black` è la "
"scelta abituale. `noise` e `pseudorandom` scoraggiano entrambi le "
"superfici semitrasparenti facendo cadere un pixel scoperto su un "
"colore che cambia a ogni passo; pseudorandom disegna piastrelle "
"vivaci invece di grana per pixel, che la perdita non può mediare, "
"e quindi preme di più. `sh` impara un cielo perché lo sfondo "
"lontano sia rappresentato invece che ignorato."),
NL("Wat pixels vult die geen splat bedekt. `black` is de gebruikelijke "
"keuze. `noise` en `pseudorandom` ontmoedigen allebei "
"halfdoorzichtige oppervlakken doordat een onbedekte pixel op een "
"kleur valt die elke stap verandert; pseudorandom tekent felle "
"tegels in plaats van korrel per pixel, die het verlies niet kan "
"wegmiddelen, en drukt dus harder. `sh` leert een skybox zodat "
"verre achtergrond wordt weergegeven in plaats van genegeerd."),
RU("Чем заполняются пиксели, которые не покрыл ни один сплат. `black` "
"— обычный выбор. `noise` и `pseudorandom` оба мешают "
"полупрозрачным поверхностям: непокрытый пиксель попадает на цвет, "
"меняющийся каждый шаг; pseudorandom рисует яркие плитки, а не "
"зерно на каждый пиксель, и функция потерь не может их усреднить, "
"поэтому давит сильнее. `sh` обучает скайбокс, чтобы дальний фон "
"был представлен, а не проигнорирован."),
TR("Hiçbir splat'ın kaplamadığı pikselleri neyin dolduracağı. `black` "
"olağan seçimdir, `noise` arka planın saydamlaşmasını caydırır, `sh` "
"ise bir gökyüzü kutusu öğrenerek uzak arka planın yok sayılmak yerine "
"temsil edilmesini sağlar."));
"alışılmış seçimdir. `noise` ve `pseudorandom` yarı saydam "
"yüzeyleri caydırır: kaplanmamış bir piksel her adımda değişen bir "
"renge düşer; pseudorandom piksel başına tanecik yerine canlı "
"karolar çizer, kayıp bunları ortalamayla silemez, bu yüzden daha "
"çok bastırır. `sh` bir gökyüzü öğrenir, böylece uzak arka plan yok "
"sayılmak yerine temsil edilir."));
SS_MSG(background_sh_degree,
EN("Skybox detail"), JA("スカイボックスの細かさ"), ZH_HANS("天空盒细节"),
+60 -14
View File
@@ -128,12 +128,63 @@ void blend_background_backward(
// Blend Background with Random Noise
// ================
// Side, in pixels, of a `blocky` background tile. Per-pixel noise is averaged
// back into flat grey by SSIM's 11x11 window and by the multi-scale pyramid,
// which is where the penalty is supposed to land; a tile this size survives both.
static constexpr unsigned kBgBlockPx = 64u;
// Murmur-style finalizer. Cheap, well-distributed, and stateless -- the whole
// background is computed, never stored.
__device__ __forceinline__ uint32_t _bg_mix(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// Unit sample for the background, in [-1, 1] either way. `blocky` draws one of
// the 8 RGB cube corners per tile: the extremes, so residual transparency costs
// the most it can. Runtime, not a template, to match the Vulkan param field.
__device__ __forceinline__ float3 _bg_sample(bool blocky, uint32_t seed, unsigned gid,
unsigned bid, unsigned x, unsigned y) {
float3 u;
if (blocky) {
// The whole tile grid shifts each step, so no pixel keeps its colour
// and the pattern cannot be baked into the splats.
const unsigned ox = _bg_mix(seed * 2u + 1u) % kBgBlockPx;
const unsigned oy = _bg_mix(seed * 2u + 7u) % kBgBlockPx;
const uint32_t h = _bg_mix(((x + ox) / kBgBlockPx) * 2654435761u
^ ((y + oy) / kBgBlockPx) * 40503u
^ (seed + bid * 0x9e3779b9u));
u.x = (h & 1u) ? 1.0f : -1.0f;
u.y = (h & 2u) ? 1.0f : -1.0f;
u.z = (h & 4u) ? 1.0f : -1.0f;
} else {
// 2u-1: without it the plain path lands in [0.5, 0.5+w/2) instead of
// straddling 0.5, so every channel sits in the same bright half.
u.x = (float)hash_uint3(seed + 0, gid, bid) * exp2f(-31.0f) - 1.0f;
u.y = (float)hash_uint3(seed + 1, gid, bid) * exp2f(-31.0f) - 1.0f;
u.z = (float)hash_uint3(seed + 2, gid, bid) * exp2f(-31.0f) - 1.0f;
}
return u;
}
template<int Transfer, bool IsLinear>
__device__ __forceinline__ float3 _bg_color(bool blocky, uint32_t seed, unsigned gid,
unsigned bid, unsigned x, unsigned y,
float randomize_weight) {
float3 background = _bg_sample(blocky, seed, gid, bid, x, y);
background = 0.5 + 0.5*randomize_weight * background;
return SlangPixelWise::display_to_working3(background, Transfer, IsLinear);
}
template<int Transfer, bool IsLinear>
__global__ void blend_background_noise_forward_kernel(
const TensorView<float, 4> in_rgb,
const TensorView<float, 4> in_transmittance,
const float randomize_weight,
const uint32_t seed,
const bool blocky,
TensorView<float, 4> out_rgb
) {
unsigned gid = blockIdx.x * blockDim.x + threadIdx.x;
@@ -147,12 +198,8 @@ __global__ void blend_background_noise_forward_kernel(
float3 rgb = in_rgb.load3(bid, y, x);
float transmittance = in_transmittance.load1(bid, y, x);
float3 background;
background.x = (float)hash_uint3(seed + 0, gid, bid) * exp2f(-32.0f);
background.y = (float)hash_uint3(seed + 1, gid, bid) * exp2f(-32.0f);
background.z = (float)hash_uint3(seed + 2, gid, bid) * exp2f(-32.0f);
background = 0.5 + 0.5*randomize_weight * (2.0f * background - 1.0f);
background = SlangPixelWise::display_to_working3(background, Transfer, IsLinear);
float3 background =
_bg_color<Transfer, IsLinear>(blocky, seed, gid, bid, x, y, randomize_weight);
rgb = SlangPixelWise::blend_background(rgb, transmittance, background);
@@ -165,6 +212,7 @@ __global__ void blend_background_noise_backward_kernel(
const TensorView<float, 4> in_transmittance,
const float randomize_weight,
const uint32_t seed,
const bool blocky,
const float overexposure_scale,
const TensorView<float, 4> v_out_rgb,
TensorView<float, 4> v_in_rgb,
@@ -181,12 +229,8 @@ __global__ void blend_background_noise_backward_kernel(
float3 rgb = in_rgb.load3(bid, y, x);
float transmittance = in_transmittance.load1(bid, y, x);
float3 background;
background.x = (float)hash_uint3(seed + 0, gid, bid) * exp2f(-32.0f);
background.y = (float)hash_uint3(seed + 1, gid, bid) * exp2f(-32.0f);
background.z = (float)hash_uint3(seed + 2, gid, bid) * exp2f(-32.0f);
background = 0.5 + 0.5*randomize_weight * (2.0f * background - 1.0f);
background = SlangPixelWise::display_to_working3(background, Transfer, IsLinear);
float3 background =
_bg_color<Transfer, IsLinear>(blocky, seed, gid, bid, x, y, randomize_weight);
float3 v_out = v_out_rgb.load3(bid, y, x);
@@ -205,6 +249,7 @@ __global__ void blend_background_noise_backward_kernel(
void blend_background_noise_forward(
int transfer,
bool is_linear,
bool blocky, // tiled RGB corners instead of U[0,1)
DeviceTensor3D<float3> rgb, // [B, H, W, 3]
DeviceTensor3D<float> transmittance, // [B, H, W, 1]
float randomize_weight,
@@ -216,7 +261,7 @@ void blend_background_noise_forward(
_XFER_PICK(blend_background_noise_forward_kernel, transfer, is_linear)
<<<_LAUNCH_ARGS_2D(h*w, b, 256, 1)>>>(
_dt3d_to_tv4<float>(rgb), _dt3d_to_tv4<float>(transmittance),
randomize_weight, seed,
randomize_weight, seed, blocky,
_dt3d_to_tv4<float>(out_rgb)
);
CHECK_DEVICE_ERROR(cudaGetLastError());
@@ -226,6 +271,7 @@ void blend_background_noise_forward(
void blend_background_noise_backward(
int transfer,
bool is_linear,
bool blocky, // tiled RGB corners instead of noise
DeviceTensor3D<float3> rgb, // [B, H, W, 3] PRE-blend
DeviceTensor3D<float> transmittance, // [B, H, W, 1]
float randomize_weight,
@@ -240,7 +286,7 @@ void blend_background_noise_backward(
_XFER_PICK(blend_background_noise_backward_kernel, transfer, is_linear)
<<<_LAUNCH_ARGS_2D(h*w, b, 256, 1)>>>(
_dt3d_to_tv4<float>(rgb), _dt3d_to_tv4<float>(transmittance),
randomize_weight, seed,
randomize_weight, seed, blocky,
_overexposure_scale(b, h, w, overexposure_weight),
_dt3d_to_tv4<float>(v_out_rgb),
_dt3d_to_tv4<float>(v_rgb), _dt3d_to_tv4<float>(v_transmittance)
+2
View File
@@ -205,6 +205,7 @@ void blend_background_backward(
void blend_background_noise_forward(
int transfer,
bool is_linear,
bool blocky, // tiled RGB corners instead of U[0,1)
DeviceTensor3D<float3> rgb, // [B, H, W, 3]
DeviceTensor3D<float> transmittance, // [B, H, W, 1]
float randomize_weight,
@@ -216,6 +217,7 @@ void blend_background_noise_forward(
void blend_background_noise_backward(
int transfer,
bool is_linear,
bool blocky, // tiled RGB corners instead of noise
DeviceTensor3D<float3> rgb, // [B, H, W, 3] PRE-blend
DeviceTensor3D<float> transmittance, // [B, H, W, 1]
float randomize_weight,