Skip to content

Commit 88f1070

Browse files
Merge pull request #3523 from sibocw:fix-python-ortho-cam-depth
PiperOrigin-RevId: 982434876 Change-Id: I932619e0128119d01d6b3c374b34f056cbac6e00
2 parents 95eb305 + 6a6e1d4 commit 88f1070

2 files changed

Lines changed: 73 additions & 11 deletions

File tree

‎python/mujoco/renderer_test.py‎

Lines changed: 56 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -167,6 +167,62 @@ def test_renderer_output_with_out(self):
167167
with self.assertRaises(ValueError):
168168
renderer.render(out=np.zeros((*failing_render_size, 3), np.uint8))
169169

170+
def test_renderer_depth_rendering_camera_projection(self):
171+
"""Tests orthographic vs perspective camera depth rendering projections.
172+
173+
Orthographic depth uses a linear NDC mapping (glOrtho); perspective depth
174+
uses a hyperbolic mapping (glFrustum).
175+
176+
Render the same tilted plane using identically positioned orthographic and
177+
perspective cameras. The two should agree on the depth at the center pixel
178+
(actual distance), but they should have different depth values at non-center
179+
pixels because the rays of the perspective camera diverge while those of the
180+
orthographic camera remain parallel.
181+
"""
182+
distance = 3.0
183+
ortho_fovy = 2.0 * distance * np.tan(np.deg2rad(45.0 / 2.0))
184+
xml = f"""
185+
<mujoco>
186+
<statistic extent="1"/>
187+
<visual>
188+
<quality offsamples="0"/>
189+
</visual>
190+
<worldbody>
191+
<camera name="ortho" pos="0 0 {distance}" xyaxes="1 0 0 0 1 0" projection="orthographic" fovy="{ortho_fovy}"/>
192+
<camera name="persp" pos="0 0 {distance}" xyaxes="1 0 0 0 1 0" projection="perspective" fovy="45"/>
193+
<geom name="floor" type="plane" size="5 5 0.1" zaxis="0 0.3 1"/>
194+
</worldbody>
195+
</mujoco>
196+
"""
197+
model = mujoco.MjModel.from_xml_string(xml)
198+
data = mujoco.MjData(model)
199+
mujoco.mj_forward(model, data)
200+
201+
# Odd resolution so a single pixel sits exactly on the optical axis.
202+
def render_depth(camera_name):
203+
with mujoco.Renderer(model, 51, 51) as renderer:
204+
renderer.enable_depth_rendering()
205+
renderer.update_scene(data, camera_name)
206+
return renderer.scene.camera[0].orthographic, renderer.render().copy()
207+
208+
is_ortho, ortho_depth = render_depth('ortho')
209+
is_persp, persp_depth = render_depth('persp')
210+
211+
# Check that the scenes are actually rendered with the expected cameras.
212+
self.assertTrue(is_ortho)
213+
self.assertFalse(is_persp)
214+
215+
# Center pixels of both cameras should show the actual distance
216+
center = ortho_depth.shape[0] // 2, ortho_depth.shape[1] // 2
217+
self.assertAlmostEqual(ortho_depth[center], distance, delta=1e-2)
218+
self.assertAlmostEqual(persp_depth[center], distance, delta=1e-2)
219+
220+
# Depth values at the corner should differ
221+
corner = (0, 0)
222+
self.assertNotAlmostEqual(
223+
ortho_depth[corner], persp_depth[corner], delta=1e-2
224+
)
225+
170226
def test_renderer_del_safe_when_init_fails_early(self):
171227
"""Regression test for #3213.
172228

‎python/mujoco/rendering/classic/renderer.py‎

Lines changed: 17 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -195,22 +195,28 @@ def render(self, *, out: Optional[np.ndarray] = None) -> np.ndarray:
195195
# https://registry.khronos.org/OpenGL-Refpages/gl2.1/xhtml/glFrustum.xml
196196
zfar = np.float32(far)
197197
znear = np.float32(near)
198-
c_coef = -(zfar + znear) / (zfar - znear)
199-
d_coef = -(np.float32(2) * zfar * znear) / (zfar - znear)
200-
201-
# In reverse Z mode the perspective matrix is transformed by the following
202-
c_coef = np.float32(-0.5) * c_coef - np.float32(0.5)
203-
d_coef = np.float32(-0.5) * d_coef
204198

205199
# We need 64 bits to convert Z from ndc to metric depth without noticeable
206200
# losses in precision
207201
out_64 = out.astype(np.float64)
208202

209-
# Undo OpenGL projection
210-
# Note: We do not need to take action to convert from window coordinates
211-
# to normalized device coordinates because in reversed Z mode the mapping
212-
# is identity
213-
out_64 = d_coef / (out_64 + c_coef)
203+
if self._scene.camera[0].orthographic:
204+
# Orthographic cameras: mapping is linear
205+
out_64 = zfar - out_64 * (zfar - znear)
206+
else:
207+
# Perspective cameras: mapping is hyperbolic
208+
c_coef = -(zfar + znear) / (zfar - znear)
209+
d_coef = -(np.float32(2) * zfar * znear) / (zfar - znear)
210+
211+
# In reverse Z mode the perspective matrix is transformed by
212+
c_coef = np.float32(-0.5) * c_coef - np.float32(0.5)
213+
d_coef = np.float32(-0.5) * d_coef
214+
215+
# Undo OpenGL projection
216+
# Note: We do not need to take action to convert from window coordinates
217+
# to normalized device coordinates because in reversed Z mode
218+
# the mapping is identity
219+
out_64 = d_coef / (out_64 + c_coef)
214220

215221
# Cast result back to float32 for backwards compatibility
216222
# This has a small accuracy cost

0 commit comments

Comments
 (0)