// Three.js scene (spec M5): point cloud (/api/pointcloud via PLYLoader), each video's camera // path as a line + current-pose frustum wireframe (colored per video), OrbitControls free roam, // and snap-to-camera. ALL COLMAP->Three.js conversion goes through the FROZEN lib/pose.js // helper (colmapToThreejs) — the diag(1,-1,-1) math is never reimplemented here. import * as THREE from "three"; import { OrbitControls } from "three/addons/controls/OrbitControls.js"; import { PLYLoader } from "three/addons/loaders/PLYLoader.js"; import { colmapToThreejs } from "./lib/pose.js"; import { poseAt } from "./lib/poseTrack.js"; import { tVideoFromGlobal } from "./lib/timebase.js"; import { state, emit } from "./state.js"; import { transport } from "./transport.js"; // Per-video colors (match the frustum, path line, and grid accents). export const VIDEO_COLORS = ["#ff5d73", "#4dd0ff", "#c9a2ff", "#ffcf5d", "#7dffb0", "#ff9d5d"]; export function videoColor(index) { return VIDEO_COLORS[index % VIDEO_COLORS.length]; } const FRUSTUM_DEPTH = 1.2; const DEFAULT_FOV = 50; const STAGE_TARGET = new THREE.Vector3(0, 0.5, 0); const _m4 = new THREE.Matrix4(); const _pos = new THREE.Vector3(); const _quat = new THREE.Quaternion(); const _scale = new THREE.Vector3(); const _fwd = new THREE.Vector3(); const _up = new THREE.Vector3(); /** Build a camera-facing text label sprite (used for 3D anchor labels). */ function makeLabelSprite(text) { const font = 48; const pad = 10; const c = document.createElement("canvas"); let ctx = c.getContext("2d"); ctx.font = `600 ${font}px -apple-system, system-ui, sans-serif`; const w = Math.ceil(ctx.measureText(text).width) + pad * 2; const h = font + pad * 2; c.width = w; c.height = h; ctx = c.getContext("2d"); ctx.font = `600 ${font}px -apple-system, system-ui, sans-serif`; ctx.fillStyle = "rgba(6,8,14,0.74)"; ctx.fillRect(0, 0, w, h); ctx.fillStyle = "#e6e8ee"; ctx.textBaseline = "middle"; ctx.fillText(text, pad, h / 2 + 1); const tex = new THREE.CanvasTexture(c); tex.minFilter = THREE.LinearFilter; const sprite = new THREE.Sprite( new THREE.SpriteMaterial({ map: tex, depthTest: false, transparent: true }) ); const s = 0.006; sprite.scale.set(w * s, h * s, 1); return sprite; } export class Scene3D { constructor() { this.videoIndex = {}; // id -> index (for stable colors) this.rigs = {}; // id -> { frustum, marker, pick, path } this._tween = null; this._raf = null; this.anchorGroup = null; // 3D anchor spheres + labels (M6 anchors + M8 resolved) this.pathMode = false; // when true, camPath (M9) drives the camera; we don't touch it this.helpersVisible = true; // M14 hook: grid/axes/frusta/gizmos/labels toggle this._helperObjs = []; // extra helper objects registered by other modules (camPath gizmos) } init(canvas) { this.canvas = canvas; const parent = canvas.parentElement; const w = parent.clientWidth || 1; const h = parent.clientHeight || 1; this.renderer = new THREE.WebGLRenderer({ canvas, antialias: true, alpha: false }); this.renderer.setPixelRatio(Math.min(window.devicePixelRatio || 1, 2)); this.renderer.setSize(w, h, false); this.renderer.setClearColor(0x0a0c11, 1); this.scene = new THREE.Scene(); this.camera = new THREE.PerspectiveCamera(DEFAULT_FOV, w / h, 0.05, 3000); this.camera.position.set(12, 9, 15); this.camera.lookAt(STAGE_TARGET); this.controls = new OrbitControls(this.camera, this.renderer.domElement); this.controls.enableDamping = true; this.controls.dampingFactor = 0.08; this.controls.target.copy(STAGE_TARGET); this.controls.update(); // Lights (points ignore lighting, but frusta/markers use MeshBasic so this is mostly cosmetic). this.scene.add(new THREE.AmbientLight(0xffffff, 0.9)); const grid = new THREE.GridHelper(40, 40, 0x22384a, 0x161c26); grid.position.y = 0; this.scene.add(grid); const axes = new THREE.AxesHelper(1.5); this.scene.add(axes); this._grid = grid; this._axes = axes; this.anchorGroup = new THREE.Group(); this.scene.add(this.anchorGroup); this._buildRigs(); this._loadPointCloud(); this.refreshAnchors(); this._wirePicking(); this._wireResize(parent); } /** Rebuild the 3D anchor markers from state.anchors (M6 seeded + M8 resolved). */ refreshAnchors() { if (!this.anchorGroup) return; for (const child of [...this.anchorGroup.children]) { this.anchorGroup.remove(child); child.geometry?.dispose?.(); if (child.material) { child.material.map?.dispose?.(); child.material.dispose(); } } for (const a of state.anchors) { const color = new THREE.Color(a.color || "#59d499"); const sphere = new THREE.Mesh( new THREE.SphereGeometry(0.13, 16, 16), new THREE.MeshBasicMaterial({ color }) ); sphere.position.set(a.x, a.y, a.z); this.anchorGroup.add(sphere); if (a.label) { const label = makeLabelSprite(a.label); label.position.set(a.x, a.y + 0.3, a.z); label.visible = this.helpersVisible; // photo mode (M14) may be hiding labels right now this.anchorGroup.add(label); } } } /** Expose the scene graph for auxiliary overlays (e.g. camPath gizmos M9, moment FX M15). */ addObject(obj) { this.scene?.add(obj); } removeObject(obj) { this.scene?.remove(obj); } /** Register an added object as a "helper" so setHelpersVisible governs it (camPath gizmos). */ registerHelper(obj) { this._helperObjs.push(obj); obj.visible = this.helpersVisible; } /** * M14 hook (phase-5 contract #6): show/hide every non-content visual in one call — grid, * axes, camera frusta + markers + path lines, registered gizmo groups, and anchor LABEL * sprites (anchor spheres stay: friend tags belong in photos). update() keeps enforcing * the flag on the per-video rigs, whose visibility it recomputes every frame. */ setHelpersVisible(visible) { this.helpersVisible = !!visible; if (this._grid) this._grid.visible = this.helpersVisible; if (this._axes) this._axes.visible = this.helpersVisible; for (const obj of this._helperObjs) obj.visible = this.helpersVisible; if (this.anchorGroup) for (const child of this.anchorGroup.children) if (child.isSprite) child.visible = this.helpersVisible; } /** * M13 hook (phase-5 contract #6): fly the free-roam camera's attention to a world point * (anchor jump-to). Detaches follow-cam / path mode, retargets OrbitControls, and backs the * camera off along its current viewing direction. */ focusOn(x, y, z, dist = 6) { if (this.pathMode) this.exitPathMode(); if (state.followCameraId != null) { state.followCameraId = null; this._tween = null; emit("follow", null); } this.camera.fov = DEFAULT_FOV; this.camera.updateProjectionMatrix(); const dir = this.camera.position.clone().sub(this.controls.target); if (dir.lengthSq() < 1e-6) dir.set(0, 0.5, 1); dir.normalize().multiplyScalar(dist); this.controls.target.set(x, y, z); this.camera.position.set(x + dir.x, y + dir.y, z + dir.z); this.controls.enabled = true; this.controls.update(); } /** M9: hand camera control to camPath. update() then leaves the camera alone. */ enterPathMode() { this.pathMode = true; this.controls.enabled = false; state.followCameraId = null; this._tween = null; emit("follow", null); } exitPathMode() { if (!this.pathMode) return; this.pathMode = false; this.controls.target.copy(STAGE_TARGET); this.controls.enabled = true; this.controls.update(); } /** M9: set the camera pose directly (called by camPath each frame while path-playing). */ applyCameraPose(pos, quat, fov) { this.camera.position.copy(pos); this.camera.quaternion.copy(quat); if (fov != null && Math.abs(this.camera.fov - fov) > 1e-3) { this.camera.fov = fov; this.camera.updateProjectionMatrix(); } } _buildRigs() { state.videos.forEach((v, i) => { this.videoIndex[v.id] = i; const color = new THREE.Color(videoColor(i)); const group = new THREE.Group(); // Frustum wireframe (built in Three.js camera-local space: looks down -z, +y up). const poses = state.poses[v.id] || []; const intr = poses[0]?.intrinsics; const W = v.width; const H = v.height; const hx = intr ? (FRUSTUM_DEPTH * W) / (2 * intr.fx) : FRUSTUM_DEPTH * 0.6; const hy = intr ? (FRUSTUM_DEPTH * H) / (2 * intr.fy) : FRUSTUM_DEPTH * 0.34; const d = FRUSTUM_DEPTH; const c = [ [-hx, -hy, -d], [hx, -hy, -d], [hx, hy, -d], [-hx, hy, -d], ]; const apex = [0, 0, 0]; const segs = [ apex, c[0], apex, c[1], apex, c[2], apex, c[3], c[0], c[1], c[1], c[2], c[2], c[3], c[3], c[0], ]; const fg = new THREE.BufferGeometry(); fg.setAttribute("position", new THREE.Float32BufferAttribute(segs.flat(), 3)); const frustum = new THREE.LineSegments( fg, new THREE.LineBasicMaterial({ color }) ); frustum.matrixAutoUpdate = true; group.add(frustum); // Camera-center marker (visible) + an invisible larger sphere for click picking. const marker = new THREE.Mesh( new THREE.SphereGeometry(0.12, 12, 12), new THREE.MeshBasicMaterial({ color }) ); group.add(marker); this.scene.add(group); const pick = new THREE.Mesh( new THREE.SphereGeometry(0.55, 8, 8), new THREE.MeshBasicMaterial({ visible: false }) ); pick.userData.videoId = v.id; group.add(pick); // Camera path line (one vertex per stored pose center). let path = null; if (poses.length > 1) { const pts = []; for (const p of poses) { const { position } = colmapToThreejs(p.q, p.t); pts.push(position[0], position[1], position[2]); } const pg = new THREE.BufferGeometry(); pg.setAttribute("position", new THREE.Float32BufferAttribute(pts, 3)); path = new THREE.Line( pg, new THREE.LineBasicMaterial({ color, transparent: true, opacity: 0.5 }) ); this.scene.add(path); } this.rigs[v.id] = { group, frustum, marker, pick, path }; }); } _loadPointCloud() { // Prefer a trained 3DGS splat when the backend has one (docs/modelbeast-crossover.md); // fall back to the sparse COLMAP point cloud. The splat library is heavy, so it's // lazy-imported only when actually needed. if (state.hasSplat) { this._loadSplat(); return; } const loader = new PLYLoader(); loader.load( state.apiBase + "/api/pointcloud", (geometry) => { geometry.computeBoundingBox(); const hasColor = !!geometry.getAttribute("color"); const material = new THREE.PointsMaterial({ size: 0.05, sizeAttenuation: true, vertexColors: hasColor, color: hasColor ? 0xffffff : 0x88ccff, }); this.points = new THREE.Points(geometry, material); this.scene.add(this.points); emit("pointcloud-loaded", geometry.getAttribute("position")?.count ?? 0); }, undefined, (err) => { console.warn("[scene3d] point cloud load failed", err); } ); } async _loadSplat() { try { const GS = await import("@mkkellogg/gaussian-splats-3d"); // DropInViewer is a THREE.Object3D — it joins the existing scene and // renders with our camera/controls, so rigs/anchors/paths overlay as usual. const splatViewer = new GS.DropInViewer({ sharedMemoryForWorkers: false }); await splatViewer.addSplatScene(state.apiBase + "/api/splat", { showLoadingUI: false, progressiveLoad: true, format: GS.SceneFormat.Ply, }); this.splat = splatViewer; this.scene.add(splatViewer); emit("pointcloud-loaded", -1); // -1 = splat (no discrete point count) } catch (err) { console.warn("[scene3d] splat load failed — falling back to point cloud", err); state.hasSplat = false; this._loadPointCloud(); } } // Live pose (COLMAP convention) for a video at the current master time, or null. _livePose(videoId) { const poses = state.poses[videoId]; if (!poses || poses.length === 0) return null; const meta = state.videos.find((v) => v.id === videoId); const tv = tVideoFromGlobal(state.tGlobal, meta.offset_ms || 0, meta.drift_ppm || 0); return poseAt(poses, tv); } /** Snap the viewer camera to a video's pose and follow it while playing. */ snapTo(videoId) { const pose = this._livePose(videoId); if (!pose) return; if (this.pathMode) this.exitPathMode(); state.followCameraId = videoId; this.controls.enabled = false; const targetFov = (2 * Math.atan(state.videos.find((v) => v.id === videoId).height / (2 * pose.intrinsics.fy)) * 180) / Math.PI; this._tween = { start: performance.now(), dur: 600, fromPos: this.camera.position.clone(), fromQuat: this.camera.quaternion.clone(), fromFov: this.camera.fov, toFov: targetFov, }; emit("follow", videoId); } /** Detach to free roam (OrbitControls). */ freeRoam() { if (state.followCameraId == null) return; state.followCameraId = null; this._tween = null; this.camera.fov = DEFAULT_FOV; this.camera.updateProjectionMatrix(); // Re-anchor OrbitControls to the stage, orbiting from the current position. this.controls.target.copy(STAGE_TARGET); this.controls.enabled = true; this.controls.update(); emit("follow", null); } _applyPoseToCamera(pose, targetFov) { const { matrixWorld } = colmapToThreejs(pose.q, pose.t); _m4.fromArray(matrixWorld); _m4.decompose(_pos, _quat, _scale); const t = this._tween; if (t) { const k = Math.min(1, (performance.now() - t.start) / t.dur); const e = k * k * (3 - 2 * k); // smoothstep this.camera.position.lerpVectors(t.fromPos, _pos, e); this.camera.quaternion.copy(t.fromQuat).slerp(_quat, e); this.camera.fov = t.fromFov + (t.toFov - t.fromFov) * e; this.camera.updateProjectionMatrix(); if (k >= 1) this._tween = null; } else { this.camera.position.copy(_pos); this.camera.quaternion.copy(_quat); if (Math.abs(this.camera.fov - targetFov) > 1e-3) { this.camera.fov = targetFov; this.camera.updateProjectionMatrix(); } } } /** Per-frame update: move frusta to current poses, drive follow-cam, render. */ update() { for (const v of state.videos) { const rig = this.rigs[v.id]; if (!rig) continue; const pose = this._livePose(v.id); const on = pose && state.enabled[v.id] !== false; rig.group.visible = !!on && this.helpersVisible; if (rig.path) rig.path.visible = state.enabled[v.id] !== false && this.helpersVisible; if (!on) continue; const { matrixWorld } = colmapToThreejs(pose.q, pose.t); _m4.fromArray(matrixWorld); _m4.decompose(rig.group.position, rig.group.quaternion, rig.group.scale); } if (this.pathMode) { // camPath (M9) already set the camera pose this frame; don't fight it. } else if (state.followCameraId != null) { const pose = this._livePose(state.followCameraId); if (pose) { const meta = state.videos.find((v) => v.id === state.followCameraId); const targetFov = (2 * Math.atan(meta.height / (2 * pose.intrinsics.fy)) * 180) / Math.PI; this._applyPoseToCamera(pose, targetFov); } } else { this.controls.update(); } this._updateAudio(); this.renderer.render(this.scene, this.camera); } // 🎧 Feed the transport's spatial audio: listener = viewer camera, emitters = camera rigs // (rig.group already carries each camera's live pose from the loop above). _updateAudio() { if (!state.spatialAudio) return; this.camera.getWorldPosition(_pos); _fwd.set(0, 0, -1).applyQuaternion(this.camera.quaternion); _up.set(0, 1, 0).applyQuaternion(this.camera.quaternion); const positions = {}; for (const v of state.videos) { const rig = this.rigs[v.id]; if (rig?.group.visible) positions[v.id] = rig.group.position; } transport.updateSpatial( { px: _pos.x, py: _pos.y, pz: _pos.z, fx: _fwd.x, fy: _fwd.y, fz: _fwd.z, ux: _up.x, uy: _up.y, uz: _up.z }, positions ); } _wirePicking() { const el = this.renderer.domElement; const ray = new THREE.Raycaster(); const ndc = new THREE.Vector2(); let downX = 0; let downY = 0; el.addEventListener("pointerdown", (e) => { downX = e.clientX; downY = e.clientY; }); el.addEventListener("pointerup", (e) => { if (Math.hypot(e.clientX - downX, e.clientY - downY) > 5) return; // was a drag, not a click const rect = el.getBoundingClientRect(); ndc.x = ((e.clientX - rect.left) / rect.width) * 2 - 1; ndc.y = -((e.clientY - rect.top) / rect.height) * 2 + 1; ray.setFromCamera(ndc, this.camera); const picks = Object.values(this.rigs) .map((r) => r.pick) .filter((m) => m.parent && m.parent.visible); const hits = ray.intersectObjects(picks, false); if (hits.length) this.snapTo(hits[0].object.userData.videoId); }); } _wireResize(parent) { const ro = new ResizeObserver(() => { const w = parent.clientWidth || 1; const h = parent.clientHeight || 1; this.renderer.setSize(w, h, false); // Aspect always follows the 3D canvas (never the video) so nothing stretches; snap-to- // camera only overrides the vertical fov + pose, matching the video's vertical framing. this.camera.aspect = w / h; this.camera.updateProjectionMatrix(); }); ro.observe(parent); } } export const scene3d = new Scene3D();