"""Paired depth-only translation ablation; fixed hand articulation and object trajectory.""" import os os.environ.setdefault('OMP_NUM_THREADS','4') import sys,json from pathlib import Path ROOT=Path(__file__).resolve().parents[1];sys.path.insert(0,str(ROOT));sys.path.insert(0,str(ROOT/'third_party/FoundationPose')) import numpy as np,cv2,torch,trimesh,open3d as o3d,nvdiffrast.torch as dr from Utils import nvdiffrast_render,make_mesh_tensors from utils.mano_utils import MANOForwardKinematics from red_box_distance import signed_distance OUT=ROOT/'output/depth_ablation_red_20260916';SRC=ROOT/'docs/20260916_104026' torch.set_num_threads(4) meta=json.loads((SRC/'intrinsics.json').read_text());K=np.array([[meta['fx'],0,meta['cx']],[0,meta['fy'],meta['cy']],[0,0,1.]]) obj=np.load(ROOT/'output/foundationpose_red_20260916/poses.npz');P=obj['T_camera_from_object'][:315];objvalid=obj['accepted'][:315] mesh=trimesh.load(OUT/'red_box_closed.ply',process=False);assert mesh.is_watertight scene=o3d.t.geometry.RaycastingScene();scene.add_triangles(o3d.t.geometry.TriangleMesh.from_legacy(o3d.geometry.TriangleMesh(o3d.utility.Vector3dVector(mesh.vertices),o3d.utility.Vector3iVector(mesh.faces)))) ctx=dr.RasterizeCudaContext();eye=torch.eye(4,device='cuda')[None] fk=MANOForwardKinematics(str(ROOT/'third_party/hamer/_DATA/data/mano'),torch.device('cpu')) yy,xx=np.indices((480,848));train=((xx//8+yy//8)%2)==0 all_data={};report={'design':'Same RGB-only HandFlow output, fixed STL/object pose, articulation/shape unchanged. Single robust camera-ray translation using depth. No added smoothing or interpolation. Fit and evaluation use disjoint 8x8 image checkerboard cells.','limits':['Visible skin selected with a color heuristic; correspondence and occlusion errors remain.','Held-out cells belong to the same depth sensor and frame, not independent ground truth.','Vertex signed distances do not detect every triangle-triangle collision.','Contact labels and force ground truth unavailable; nearest fingertip gap is not proof of contact.'],'hands':{}} def render(v,faces): mt={'pos':torch.as_tensor(v,device='cuda',dtype=torch.float32),'faces':torch.as_tensor(faces,device='cuda',dtype=torch.int32),'vnormals':torch.zeros((len(v),3),device='cuda'),'vertex_color':torch.ones((len(v),3),device='cuda')} with torch.inference_mode():_,d,_=nvdiffrast_render(K=K,H=480,W=848,ob_in_cams=eye,glctx=ctx,mesh_tensors=mt) return d[0].cpu().numpy() def signed(v,T): local=(v-T[:3,3])@T[:3,:3] return signed_distance(scene,mesh,local) def stat(x): x=np.asarray(x);x=x[np.isfinite(x)];return {'median':float(np.median(x)),'p95':float(np.percentile(x,95)),'mean':float(np.mean(x))} if len(x) else None for side in ['foreground_left']: src=np.load(OUT/'left_mirrored'/'handflow_results.npz');v=src['verts_cam'];n=len(v);assert n==315 pose=torch.tensor(src['pose']);betas=torch.tensor(src['betas']);trans=torch.tensor(src['trans']); if betas.ndim==1:betas=betas[None].expand(n,-1) elif len(betas)==1:betas=betas.expand(n,-1) j=fk.joints(pose,betas,trans,['right']*n).numpy()/1000 reproduced=fk.verts(pose,betas,trans,['right']*n).numpy()/1000;assert np.max(np.abs(reproduced-v))<1e-5 # Restore mirrored-right geometry to physical left; reverse face winding under reflection. v=v.copy();v[:,:,0]*=-1;j[:,:,0]*=-1 faces=np.ascontiguousarray(src['faces'][:,[0,2,1]]) corrected=v.copy();jc=j.copy();deltas=np.zeros((n,3));supported=np.zeros(n,bool);rows=[];cap=cv2.VideoCapture(str(SRC/'color.mp4')) for t in range(n): ok,b=cap.read();assert ok;z=cv2.imread(str(SRC/'depth'/f'{t:06d}.png'),-1).astype('float32')*meta['depth_scale_m'] r={'frame':t,'detected':bool(src['pred_valid'][t]),'object_accepted':bool(objvalid[t]),'depth_supported':False} if src['pred_valid'][t]: d0=render(v[t],faces);skin=cv2.inRange(cv2.cvtColor(b,cv2.COLOR_BGR2YCrCb),np.array([0,133,77]),np.array([255,173,127]))>0 hsv=cv2.cvtColor(b,cv2.COLOR_BGR2HSV);red=((hsv[:,:,0]<10)|(hsv[:,:,0]>170))&(hsv[:,:,1]>115) skin &= ~red interior=cv2.erode((d0>.1).astype('uint8'),np.ones((5,5),np.uint8))>0 mask=skin&interior&(z>.1)&(z<.85) fit=mask&train;errors=z[fit]-d0[fit];delta=float(np.median(errors)) if len(errors) else 0.;mad=float(np.median(np.abs(errors-delta))) if len(errors) else 1. accept=len(errors)>=100 and mad<.025 and abs(delta)<.25 and j[t,0,2]>.1 if accept: deltas[t]=j[t,0]/j[t,0,2]*delta;corrected[t]+=deltas[t];jc[t]+=deltas[t];supported[t]=True d1=render(corrected[t],faces) if accept else d0 test=mask&(~train)&(d1>.1) r.update(depth_supported=bool(accept),fit_pixels=int(fit.sum()),fit_mad_mm=mad*1000,shift_z_mm=float(deltas[t,2]*1000),heldout_pixels=int(test.sum()),depth_error_before_mm=float(np.median(np.abs(d0[test]-z[test]))*1000) if test.any() else None,depth_error_after_mm=float(np.median(np.abs(d1[test]-z[test]))*1000) if test.any() else None) if objvalid[t]: # Use nearest surface vertices to each MANO fingertip; preserve exactly the same vertices in A/B. tips=[4,8,12,16,20];groups=[np.argsort(np.linalg.norm(v[t]-j[t,k],axis=1))[:20] for k in tips] for label,verts in [('before',v[t]),('after',corrected[t])]: sd=signed(verts,P[t]);gap=[float(np.min(np.abs(sd[g]))*1000) for g in groups] r[label]={'penetrating_vertex_fraction':float((sd<-.001).mean()),'max_vertex_penetration_mm':float(max(0,-sd.min())*1000),'nearest_fingertip_surface_gap_mm':min(gap),'fingertip_surface_gaps_mm':gap} rows.append(r) if t%60==0:print(side,t,flush=True) cap.release();paired=np.array([r['depth_supported'] and r['object_accepted'] and 'before' in r for r in rows]);held=[r for r in rows if r['depth_supported'] and r.get('heldout_pixels',0)>=100] summary={'detected_frames':int(src['pred_valid'].sum()),'depth_supported_frames':int(supported.sum()),'paired_object_frames':int(paired.sum()),'heldout_depth_frames':len(held),'translation_norm_mm':stat(np.linalg.norm(deltas[supported],axis=1)*1000),'before':{},'after':{}} triple=paired[:-2]&paired[1:-1]&paired[2:] for label,joints in [('before',j),('after',jc)]: summary[label]['heldout_depth_error_mm']=stat([r['depth_error_'+label+'_mm'] for r in held]) for key in ['penetrating_vertex_fraction','max_vertex_penetration_mm','nearest_fingertip_surface_gap_mm']:summary[label][key]=stat([rows[t][label][key] for t in np.flatnonzero(paired)]) second=np.diff(joints[:,0],n=2,axis=0)[triple];summary[label]['wrist_second_difference_rms_mm']=float(np.sqrt(np.mean(np.sum(second**2,axis=-1)))*1000) if len(second) else None summary['jitter_valid_triples']=int(triple.sum());report['hands'][side]=summary np.savez_compressed(OUT/f'{side}_depth_comparison.npz',verts_before=v,verts_after=corrected,joints_before=j,joints_after=jc,faces=faces,translation_camera=deltas,depth_supported=supported,detected=src['pred_valid'],paired=paired,K=K,object_pose=P,object_accepted=objvalid) (OUT/f'{side}_frame_metrics.json').write_text(json.dumps(rows,indent=2)) (OUT/'comparison_metrics.json').write_text(json.dumps(report,indent=2));print(json.dumps(report,indent=2))