diff --git a/README.md b/README.md index aafdc3b1..30571128 100644 --- a/README.md +++ b/README.md @@ -99,6 +99,31 @@ On read: a PLY header comment — Brush's `comment SplatRenderMode: default | mi On write: `.ply` and `.compressed.ply` carry `comment SplatRenderMode: mip | 2dgs` (Brush's spelling, whichever form was read); `.sog` and `meta.json` carry `"model": "antialiased" | "2dgs"`; `.spz` sets its antialiased bit, and warns that it cannot represent 2DGS. A 2DGS PLY output drops the `scale_2` column again. Other output formats have nowhere to record it and drop the tag silently. Combining inputs whose models disagree warns and writes the result untagged. +### Scene camera + +A SOG can carry one optional camera in `meta.json` — a rest pose, the pinhole intrinsics and a few viewing hints — so a viewer can open the scene where it was meant to be seen without a sidecar file. It matters most for splats lifted from a single photo or stereo pair, which only look right from near the camera that took it. The block is advisory and readers that don't use it ignore it: + +```json +"camera": { + "convention": "opencv", + "rig": "camera", + "rest": { "position": [0, 0, 0], "rotation": [0, 0, 0, 1] }, + "intrinsics": { "fx": 1194.67, "fy": 1194.67, "cx": 1024, "cy": 576, "width": 2048, "height": 1152 }, + "stereo": { "baseline_m": 0.063 }, + "focus": { "point": [0, 0, 1.68], "subject_m": 2.14, "near_m": 0.73, "far_m": 66.2 } +} +``` + +`rest.position` and `rest.rotation` (camera-to-world, `[x, y, z, w]`) are in the same coordinates as the gaussians; `convention` names the camera axes (`opencv`: +x right, +y down, +z forward); `intrinsics` are in pixels of a `width` × `height` image. Everything is optional. + +On read, a `.sog`, `meta.json` or `lod-meta.json` camera is kept; on write, `.sog` and `meta.json` carry it in `meta.json`, and `lod-meta.json` carries it once at its top level. Translate, rotate and scale actions move the rest pose with the scene (and scale the `focus` and `stereo` distances); other keys in the block are passed through unchanged. Other output formats have nowhere to record it and drop it. When several inputs carry a camera, the first is kept. + +`--camera-from cameras.json[:n]` sets it from a training camera in the `cameras.json` that 3DGS trainers write next to the PLY (`position`, `rotation`, `fx`, `fy`, `width`, `height`, OpenCV axes; the principal point is taken as the image centre): + +```bash +splat-transform scene.ply --camera-from cameras.json:12 scene.sog +``` + ## Actions Actions execute in the order specified and can be repeated. Any action may appear after any input or output file: @@ -142,6 +167,9 @@ Actions execute in the order specified and can be repeated. Any action may appea --stats [text|json] Print file info, per-column statistics and the fill/overdraw ratio to stdout. Default: text --info [text|json] Print structural metadata (format, per-LOD counts, extra columns) to stdout. Default: text -m, --morton-order Reorder Gaussians by Morton code (Z-order curve) + --camera-from Record training camera n (default 0) of a 3DGS cameras.json + as the input's camera (see Scene camera). Place it after + the input the poses belong to. ``` ## CLI Options diff --git a/src/cli/index.ts b/src/cli/index.ts index 129776d9..7b0493b3 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -29,6 +29,7 @@ import { resolveSplatModel, revision, selectLod, + sogCameraFromCamerasJson, stackLods, TextRenderer, Transform, @@ -37,6 +38,7 @@ import { WorkerQueue, writeLodSource, writeSource, + withCamera, type ChunkSource, type ChunkSourceMetadata, type ProcessAction, @@ -45,6 +47,7 @@ import { type Options as LibOptions, type CollisionMeshShape, type ReadFileSystem, + type SogCamera, logger } from '../lib'; // CLI-only internals (deliberately off the public lib surface): the LOD-path @@ -144,6 +147,8 @@ const stripLodTags = (actions: CliAction[]): ProcessAction[] => { type File = { filename: string; processActions: CliAction[]; + /** Camera block from `--camera-from`, in this input's own coordinates. */ + camera?: SogCamera; }; const cliOptionsConfig = { @@ -214,7 +219,8 @@ const cliOptionsConfig = { 'tag-lod': { type: 'string', short: 'l', multiple: true }, stats: { type: 'string', multiple: true }, info: { type: 'string', multiple: true }, - 'morton-order': { type: 'boolean', short: 'm', multiple: true } + 'morton-order': { type: 'boolean', short: 'm', multiple: true }, + 'camera-from': { type: 'string', multiple: true } } as const; const stringOptionNames = new Set(Object.entries(cliOptionsConfig) @@ -822,6 +828,19 @@ const parseArguments = async () => { current.processActions.push(ffAction); break; } + case 'camera-from': { + // [:index], index defaulting to the first camera + const match = /^(.*):(\d+)$/.exec(t.value); + const path = match ? match[1] : t.value; + let cameras; + try { + cameras = JSON.parse(await pathReadFile(path, 'utf-8')); + } catch (e) { + throw new Error(`Failed to read cameras JSON file: ${path}`); + } + current.camera = sogCameraFromCamerasJson(cameras, match ? parseInteger(match[2]) : 0); + break; + } } } } @@ -874,6 +893,8 @@ ACTIONS (executed in order; can be repeated) --stats [text|json] Print file info, per-column statistics and the fill/overdraw ratio to stdout. Default: text --info [text|json] Print structural metadata (format, per-LOD counts, extra columns) to stdout. Default: text -m, --morton-order Reorder Gaussians by Morton code (Z-order curve) + --camera-from Record training camera n (default 0) of a 3DGS cameras.json as the + input's camera; written to .sog / meta.json / lod-meta.json GENERAL -h, --help Show this help and exit @@ -1137,6 +1158,10 @@ const main = async () => { const outputFilename = resolve(outputArg.filename); + if (outputArg.camera) { + failExit('--camera-from applies to an input file: place it after the input whose poses it holds.'); + } + // Check for null output (discard file writing) const isNullOutput = outputArg.filename.toLowerCase() === 'null'; @@ -1289,7 +1314,8 @@ const main = async () => { }); const readFilename = fmt === 'mjs' ? `file://${inFile}` : inFile; const srcs = await readFile({ filename: readFilename, inputFormat: fmt, options: { ...options, lodSelect: [] }, params, fileSystem }); - return srcs.length === 1 ? srcs[0] : concatSource(srcs, pool); + const src = srcs.length === 1 ? srcs[0] : concatSource(srcs, pool); + return inputArg.camera ? withCamera(src, inputArg.camera) : src; }; // Stitch inputs: uniform layout -> concatSource (transforms unified as @@ -1312,12 +1338,14 @@ const main = async () => { const seen = [...new Set(sources.map(s => s.meta.model))].join(', '); logger.warn(`mixed splat models (${seen}); writing the result as '${model}'`); } + const withCam = sources.find(s => s.meta.camera); const dts: DataTable[] = []; for (const s of sources) { dts.push(await materializeToDataTable(s, pool)); await s.close(); } - return dataTableToChunkSource(combine(dts), pool.chunkSize, undefined, model); + const combinedSource = dataTableToChunkSource(combine(dts), pool.chunkSize, undefined, model); + return withCam ? withCamera(combinedSource, withCam.meta.camera, withCam.meta.transform) : combinedSource; }; const phase = logger.group(`Output ${outputArg.filename}`, { index: phaseTotal, total: phaseTotal }); @@ -1416,11 +1444,12 @@ const main = async () => { // Intrinsic multi-LOD: view each level with selectLod (shared parent); // env fetched separately. The input's own actions apply per level. const { filename: inFile, fileSystem } = resolveInput(inputArgs[0].filename); - const multi = single === 'lcc2' ? + const lodSource = single === 'lcc2' ? await readLcc2Source(fileSystem, inFile, { ...options, lodSelect: [] }, pool) : single === 'lod' ? await readLodSource(fileSystem, inFile, { ...options, lodSelect: [] }, pool) : await readLccSource(fileSystem, inFile, { ...options, lodSelect: [] }, pool); + const multi = inputArgs[0].camera ? withCamera(lodSource, inputArgs[0].camera) : lodSource; container = multi; envSource = single === 'lcc2' ? await readLcc2EnvironmentSource(fileSystem, inFile, pool) : @@ -1445,8 +1474,9 @@ const main = async () => { } const opened = await Promise.all(tagged.map(async (t) => { const { filename: inFile, fileSystem } = resolveInput(t.arg.filename); + const ply = await readPly(await fileSystem.createSource(inFile), pool); const src = await processSourceBridged( - await readPly(await fileSystem.createSource(inFile), pool), + t.arg.camera ? withCamera(ply, t.arg.camera) : ply, t.rest, pool, processOptions ); return { src, tag: t.tag }; diff --git a/src/lib/chunk/source.ts b/src/lib/chunk/source.ts index 5497ae64..b2686e71 100644 --- a/src/lib/chunk/source.ts +++ b/src/lib/chunk/source.ts @@ -1,3 +1,4 @@ +import { type SogCamera } from '../sog-camera'; import { type SplatModel } from '../splat-model'; import { type Transform } from '../utils'; import { type ChunkData } from './data'; @@ -38,6 +39,11 @@ type ChunkSourceMetadata = { readonly availableLayers: ReadonlySet; /** Per-layer stride + field map. Keyed by layer; only present for available layers. */ readonly layouts: Readonly>>; + /** + * Optional capture / rest camera (SOG `meta.json` `camera`). Like the data, + * it is stored raw: `transform` gives its meaning, and baking moves it along. + */ + readonly camera?: SogCamera; }; /** diff --git a/src/lib/decimate-uniform/decimate-source.ts b/src/lib/decimate-uniform/decimate-source.ts index 5e99503b..93f9f430 100644 --- a/src/lib/decimate-uniform/decimate-source.ts +++ b/src/lib/decimate-uniform/decimate-source.ts @@ -250,6 +250,7 @@ const decimateSource = async ( numChunks: [Math.ceil(outCount / src.meta.chunkSize)], shBands: src.meta.shBands, model: src.meta.model, + camera: src.meta.camera, extraColumns: src.meta.extraColumns, transform: src.meta.transform, availableLayers: src.meta.availableLayers, diff --git a/src/lib/decimate/decimate-source.ts b/src/lib/decimate/decimate-source.ts index 2fc549cd..6e98d4f2 100644 --- a/src/lib/decimate/decimate-source.ts +++ b/src/lib/decimate/decimate-source.ts @@ -433,6 +433,7 @@ const decimateSource = async ( numChunks: [Math.ceil(outCount / src.meta.chunkSize)], shBands: src.meta.shBands, model: src.meta.model, + camera: src.meta.camera, extraColumns: src.meta.extraColumns, transform: src.meta.transform, availableLayers: src.meta.availableLayers, diff --git a/src/lib/index.ts b/src/lib/index.ts index 7a454da1..c54d5939 100644 --- a/src/lib/index.ts +++ b/src/lib/index.ts @@ -12,13 +12,18 @@ export type { } from './chunk'; // Structural combinators (lazy views over sources) -export { bakeTransform, concatSource, selectLod, stackLods, sortMortonColumns, sortMortonInterleaved } from './ops'; +export { bakeTransform, concatSource, selectLod, stackLods, sortMortonColumns, sortMortonInterleaved, withCamera } from './ops'; // How a scene was trained (carried on `ChunkSourceMetadata.model`). The per-format // spellings of the tag live with their reader/writer. export { isSplatModel, resolveSplatModel } from './splat-model'; export type { SplatModel } from './splat-model'; +// The optional camera block of SOG meta.json / lod-meta.json (carried on +// `ChunkSourceMetadata.camera`). +export { sogCameraFromCamerasJson, transformSogCamera } from './sog-camera'; +export type { SogCamera } from './sog-camera'; + // Action processing over a source: `processSource` streams and throws on // actions that need the DataTable bridge; `processSourceBridged` handles every // action, materializing only the DataTable-only runs as islands. diff --git a/src/lib/ops/bake-transform.ts b/src/lib/ops/bake-transform.ts index 59e57c88..dc43600b 100644 --- a/src/lib/ops/bake-transform.ts +++ b/src/lib/ops/bake-transform.ts @@ -7,6 +7,7 @@ import { type ChunkSourceMetadata, SH_REST_COUNTS } from '../chunk'; +import { transformSogCamera } from '../sog-camera'; import { RotateSH, Transform } from '../utils'; const SH_PER_CHANNEL = [0, 3, 8, 15]; @@ -33,13 +34,19 @@ const SH_PER_CHANNEL = [0, 3, 8, 15]; * @returns A derived source whose reads yield data in `targetSpace`. */ const bakeTransform = (src: ChunkSource, targetSpace: Transform): ChunkSource => { - const meta: ChunkSourceMetadata = { ...src.meta, transform: targetSpace.clone() }; const delta = targetSpace.clone().invert().mul(src.meta.transform); if (delta.isIdentity()) { + const meta: ChunkSourceMetadata = { ...src.meta, transform: targetSpace.clone() }; return { meta, read: req => src.read(req), close: () => src.close() }; } + const meta: ChunkSourceMetadata = { + ...src.meta, + transform: targetSpace.clone(), + ...(src.meta.camera ? { camera: transformSogCamera(src.meta.camera, delta) } : {}) + }; + const r = delta.rotation; const s = delta.scale; const rx = r.x, ry = r.y, rz = r.z, rw = r.w; diff --git a/src/lib/ops/concat-source.ts b/src/lib/ops/concat-source.ts index f8720422..db598aac 100644 --- a/src/lib/ops/concat-source.ts +++ b/src/lib/ops/concat-source.ts @@ -87,6 +87,14 @@ const concatSource = (allSources: ChunkSource[], pool: ChunkDataPool): ChunkSour logger.warn(`mixed splat models (${seen}); writing the result as '${model}'`); } + // One output holds one camera: keep the first input's (the inputs share a + // transform, so it needs no re-expressing), and say so if another disagreed. + const cameras = allSources.map(s => s.meta.camera).filter(c => c !== undefined); + const camera = cameras[0]; + if (cameras.some(c => JSON.stringify(c) !== JSON.stringify(camera))) { + logger.warn('inputs carry different cameras; keeping the first'); + } + const S = ref.chunkSize; // Per-source gaussian counts and the output-row offset each source begins at. const counts = sources.map(s => s.meta.numGaussians); @@ -100,6 +108,7 @@ const concatSource = (allSources: ChunkSource[], pool: ChunkDataPool): ChunkSour const meta: ChunkSourceMetadata = { ...ref, model, + camera, numGaussians: total, numLods: 1, lodCounts: [total], diff --git a/src/lib/ops/index.ts b/src/lib/ops/index.ts index e487b3f1..87cc87ff 100644 --- a/src/lib/ops/index.ts +++ b/src/lib/ops/index.ts @@ -8,6 +8,7 @@ export { selectLod, resolveLodLevels } from './select-lod'; export { filterSource } from './filter-source'; export { reduceBandsSource } from './reduce-bands-source'; export { concatSource } from './concat-source'; +export { withCamera } from './with-camera'; export { filterNaNRows, filterByValueRows, filterBoxRows, filterSphereRows } from './filter-mask'; export { computeSourceStats } from './stats'; export type { LodStats, LodStatsData, SourceStats } from './stats'; diff --git a/src/lib/ops/with-camera.ts b/src/lib/ops/with-camera.ts new file mode 100644 index 00000000..6c30fc08 --- /dev/null +++ b/src/lib/ops/with-camera.ts @@ -0,0 +1,32 @@ +import { type ChunkSource, type ChunkSourceMetadata } from '../chunk'; +import { type SogCamera, transformSogCamera } from '../sog-camera'; +import { type Transform } from '../utils'; + +/** + * Set (or, with `undefined`, clear) a source's camera block, lazily. Reads pass + * through unchanged. + * + * `space` is the pending transform the camera's values were stored under — by + * default the source's own, i.e. the camera is in the same raw coordinates as + * the source's gaussians. When it differs (the source was re-bridged or baked + * since), the camera is re-expressed so that it still means the same pose. + * + * @param src - The parent source. + * @param camera - The camera block, or `undefined` to drop it. + * @param space - The pending transform `camera` is stored under. Default: `src.meta.transform`. + * @returns A derived source carrying the camera. + */ +const withCamera = (src: ChunkSource, camera: SogCamera | undefined, space?: Transform): ChunkSource => { + const { transform } = src.meta; + const cam = camera && space && !space.equals(transform) ? + transformSogCamera(camera, transform.clone().invert().mul(space)) : + camera; + const meta: ChunkSourceMetadata = { ...src.meta, camera: cam }; + return { + meta, + read: req => src.read(req), + close: () => src.close() + }; +}; + +export { withCamera }; diff --git a/src/lib/process-source.ts b/src/lib/process-source.ts index 9cd1cd4e..af8f5647 100644 --- a/src/lib/process-source.ts +++ b/src/lib/process-source.ts @@ -10,7 +10,8 @@ import { mapSource, mortonOrder, permuteSource, - reduceBandsSource + reduceBandsSource, + withCamera } from './ops'; import { processDataTable, type ProcessAction, type ProcessOptions } from './process'; import { formatSourceInfo, formatSourceStats } from './source-info'; @@ -151,9 +152,11 @@ const processSourceBridged = async ( } else { // DataTable island: materialize the current (streaming) source, apply // the run on the table, and re-bridge back to a source to keep going. + const { camera, transform } = src.meta; const dt = await materializeToDataTable(src, pool); await src.close(); src = dataTableToChunkSource(await processDataTable(dt, run, options), pool.chunkSize); + if (camera) src = withCamera(src, camera, transform); } i = j; } diff --git a/src/lib/readers/read-lod.ts b/src/lib/readers/read-lod.ts index 384f2309..bf1014da 100644 --- a/src/lib/readers/read-lod.ts +++ b/src/lib/readers/read-lod.ts @@ -2,6 +2,8 @@ import { containerSource, type ContainerSegment } from './container-source'; import { readSogSource } from './read-sog'; import { type ChunkDataPool, type ChunkSource } from '../chunk'; import { dirname, join, readFile, type ReadFileSystem } from '../io/read'; +import { withCamera } from '../ops'; +import { readSogCamera, type SogCamera } from '../sog-camera'; import { type Options } from '../types'; type LodReference = { @@ -22,6 +24,7 @@ type LodMeta = { counts?: number[]; lodLevels: number; environment?: string; + camera?: SogCamera; filenames: string[]; tree: LodNode; }; @@ -155,7 +158,9 @@ const readLodSource = async ( }); }); - return containerSource(segmentsByLod, pool); + const source = await containerSource(segmentsByLod, pool); + const camera = readSogCamera(meta.camera); + return camera ? withCamera(source, camera) : source; }; /** diff --git a/src/lib/readers/read-sog.ts b/src/lib/readers/read-sog.ts index a648fc88..b99f91da 100644 --- a/src/lib/readers/read-sog.ts +++ b/src/lib/readers/read-sog.ts @@ -17,6 +17,7 @@ import { import { dataTableToChunkSource, materializeToDataTable } from '../compat/data-table'; import { type DataTable } from '../data-table'; import { basename, dirname, join, type ReadFileSystem, readFile } from '../io/read'; +import { readSogCamera, type SogCamera } from '../sog-camera'; import { isSplatModel, type SplatModel } from '../splat-model'; import { logger, Transform, WebPCodec } from '../utils'; import { readSogV1, type MetaV1 } from './read-sog-v1'; @@ -33,6 +34,7 @@ type MetaV2 = { quats: { files: string[] }; sh0: { codebook: number[]; files: string[] }; shN?: { count: number; bands: number; codebook: number[]; files: string[] }; + camera?: SogCamera; }; type ReadSogOptions = { @@ -190,6 +192,7 @@ const readSogSourceV2 = async ( color: { stride: colorStride(shBands), fields: colorFields(shBands) } }; + const camera = readSogCamera(meta.camera); const meta_: ChunkSourceMetadata = { numGaussians: count, numLods: 1, @@ -201,7 +204,8 @@ const readSogSourceV2 = async ( extraColumns: [], transform: Transform.PLY.clone(), availableLayers: new Set(['position', 'geometric', 'color']), - layouts + layouts, + ...(camera ? { camera } : {}) }; // Expand source gaussian `g` into output row `r` of the requested layers. diff --git a/src/lib/sog-camera.ts b/src/lib/sog-camera.ts new file mode 100644 index 00000000..2d2e6851 --- /dev/null +++ b/src/lib/sog-camera.ts @@ -0,0 +1,163 @@ +import { Quat, Vec3 } from 'playcanvas'; + +import { logger, type Transform } from './utils'; + +/** + * The optional `camera` block of a SOG `meta.json` (and of a streamed SOG's + * `lod-meta.json`): one camera that says how the scene is meant to be opened. + * Purely advisory — a reader that doesn't want it ignores it. + * + * Positions and distances are in the same coordinates as the file's gaussians + * (scene units; metres for a metric capture). Every field is optional, and keys + * this type doesn't list are carried through untouched. + */ +type SogCamera = { + /** + * The camera's local axes. `opencv` (+x right, +y down, +z forward) is what + * COLMAP and 3DGS trainers use. + */ + convention?: string; + /** + * The intended framing: `camera` (open at the rest camera and keep its + * frustum) or `display` (orbit the subject; `rest` is the initial viewpoint). + */ + rig?: string; + /** The rest pose: camera centre, and camera-to-world rotation as `[x, y, z, w]`. */ + rest?: { position?: number[]; rotation?: number[] }; + /** Pinhole intrinsics in pixels of a `width` × `height` image. */ + intrinsics?: { fx?: number; fy?: number; cx?: number; cy?: number; width?: number; height?: number }; + /** A stereo capture's second eye sits at `+x · baseline_m` in the rest camera's frame. */ + stereo?: { baseline_m?: number }; + /** + * Scene facts a viewer can't recover from the geometry: the focus / orbit + * point (in the rest camera's frame), the subject depth and a depth range. + */ + focus?: { point?: number[]; subject_m?: number; near_m?: number; far_m?: number }; +}; + +const isObject = (v: unknown): v is Record => typeof v === 'object' && v !== null && !Array.isArray(v); +const isVec = (v: unknown, n: number): v is number[] => Array.isArray(v) && v.length === n && v.every(Number.isFinite); + +/** + * Validate a `camera` value read from a meta file, returning a deep copy (so a + * later transform never writes into the parsed document), or `undefined` with a + * warning when it isn't an object. + * + * @param value - The parsed `camera` value. + * @returns The camera block, or `undefined`. + * @ignore + */ +const readSogCamera = (value: unknown): SogCamera | undefined => { + if (value === undefined) return undefined; + if (!isObject(value)) { + logger.warn('ignoring meta.json \'camera\': expected an object'); + return undefined; + } + return structuredClone(value) as SogCamera; +}; + +/** + * Re-express a camera block under a coordinate-space transform: the rest pose + * gets the full transform, and scene distances (`focus`, `stereo`) the uniform + * scale. Intrinsics and unknown keys are unchanged. Returns a new object. + * + * @param camera - The camera block. + * @param transform - The transform to apply. + * @returns The transformed camera block. + */ +const transformSogCamera = (camera: SogCamera, transform: Transform): SogCamera => { + const result = structuredClone(camera); + if (transform.isIdentity()) return result; + + const { rest, stereo, focus } = result; + if (rest && isVec(rest.position, 3)) { + const p = transform.transformPoint(new Vec3(rest.position[0], rest.position[1], rest.position[2]), new Vec3()); + rest.position = [p.x, p.y, p.z]; + } + if (rest && isVec(rest.rotation, 4)) { + const [x, y, z, w] = rest.rotation; + const q = new Quat().mul2(transform.rotation, new Quat(x, y, z, w)).normalize(); + rest.rotation = [q.x, q.y, q.z, q.w]; + } + + const s = transform.scale; + if (s !== 1) { + if (stereo && Number.isFinite(stereo.baseline_m)) stereo.baseline_m *= s; + if (focus) { + if (isVec(focus.point, 3)) focus.point = focus.point.map(v => v * s); + if (Number.isFinite(focus.subject_m)) focus.subject_m *= s; + if (Number.isFinite(focus.near_m)) focus.near_m *= s; + if (Number.isFinite(focus.far_m)) focus.far_m *= s; + } + } + return result; +}; + +// Rotation matrix given as rows -> unit quaternion [x, y, z, w]. +const quatFromRows = (m: number[][]): number[] => { + const [[m00, m01, m02], [m10, m11, m12], [m20, m21, m22]] = m; + const trace = m00 + m11 + m22; + let x, y, z, w; + if (trace > 0) { + const s = 0.5 / Math.sqrt(trace + 1); + w = 0.25 / s; + x = (m21 - m12) * s; + y = (m02 - m20) * s; + z = (m10 - m01) * s; + } else if (m00 > m11 && m00 > m22) { + const s = 2 * Math.sqrt(1 + m00 - m11 - m22); + w = (m21 - m12) / s; + x = 0.25 * s; + y = (m01 + m10) / s; + z = (m02 + m20) / s; + } else if (m11 > m22) { + const s = 2 * Math.sqrt(1 + m11 - m00 - m22); + w = (m02 - m20) / s; + x = (m01 + m10) / s; + y = 0.25 * s; + z = (m12 + m21) / s; + } else { + const s = 2 * Math.sqrt(1 + m22 - m00 - m11); + w = (m10 - m01) / s; + x = (m02 + m20) / s; + y = (m12 + m21) / s; + z = 0.25 * s; + } + const len = Math.hypot(x, y, z, w); + return [x / len, y / len, z / len, w / len]; +}; + +/** + * Build a camera block from one entry of a 3DGS training `cameras.json` (the + * INRIA layout: `position`, a camera-to-world `rotation` matrix given as rows, + * pixel `fx`/`fy`, `width`/`height`). Those poses are in the trained PLY's own + * coordinates with OpenCV camera axes, so they are used as-is. The file has no + * principal point, so it is taken as the image centre. + * + * @param cameras - The parsed `cameras.json` (an array of cameras). + * @param index - Which camera to use. Default: 0. + * @returns The camera block. + * @throws If the entry is missing or malformed. + */ +const sogCameraFromCamerasJson = (cameras: unknown, index = 0): SogCamera => { + if (!Array.isArray(cameras)) { + throw new Error('cameras.json: expected an array of cameras'); + } + const cam = cameras[index]; + if (!isObject(cam)) { + throw new Error(`cameras.json: no camera at index ${index} (${cameras.length} cameras)`); + } + const { position, rotation, fx, fy, width, height } = cam; + if (!isVec(position, 3) || !Array.isArray(rotation) || rotation.length !== 3 || !rotation.every(r => isVec(r, 3)) || + ![fx, fy, width, height].every(v => Number.isFinite(v) && v > 0)) { + throw new Error(`cameras.json: camera ${index} needs position, a 3x3 rotation, fx, fy, width and height`); + } + return { + convention: 'opencv', + rest: { position: [...position], rotation: quatFromRows(rotation) }, + intrinsics: { fx, fy, cx: width / 2, cy: height / 2, width, height } + }; +}; + +export { readSogCamera, sogCameraFromCamerasJson, transformSogCamera }; +export type { SogCamera }; diff --git a/src/lib/writers/write-lod.ts b/src/lib/writers/write-lod.ts index e6694b98..e9fb6b27 100644 --- a/src/lib/writers/write-lod.ts +++ b/src/lib/writers/write-lod.ts @@ -8,7 +8,8 @@ import { type ChunkDataPool, type ChunkLayer, type ChunkSource, type ReadRequest import { materializeToDataTable } from '../compat/data-table'; import { Column, DataTable } from '../data-table'; import { type FileSystem } from '../io/write'; -import { bakeTransform, permuteSource, sortMortonColumns } from '../ops'; +import { bakeTransform, permuteSource, sortMortonColumns, withCamera } from '../ops'; +import { type SogCamera } from '../sog-camera'; import { BTreeNode, BTree } from '../spatial'; import type { DeviceCreator } from '../types'; import { logger, Transform } from '../utils'; @@ -49,6 +50,8 @@ type LodMeta = { */ chunkMinGaussians: number; }; + /** The scene's camera block, when the source carried one (see SOG `meta.json`). */ + camera?: SogCamera; count: number; counts: number[]; lodLevels: number; @@ -620,6 +623,10 @@ const writeLodSource = async (options: WriteLodSourceOptions, fs: FileSystem) => // bakes it itself. const mainSource = bakeTransform(options.mainSource, Transform.PLY); + // The camera goes once, into lod-meta.json; the per-unit and env meta.json + // files don't repeat it. + const unitParent = mainSource.meta.camera ? withCamera(mainSource, undefined) : mainSource; + // Pool for slim extraction read buffers and the chunk-native SOG encodes. const pool = createChunkDataPool(); @@ -788,6 +795,7 @@ const writeLodSource = async (options: WriteLodSourceOptions, fs: FileSystem) => chunkExtent: binDim, chunkMinGaussians: binMin }, + ...(mainSource.meta.camera ? { camera: mainSource.meta.camera } : {}), count: counts.reduce((acc, curr) => acc + curr, 0), counts, lodLevels, @@ -831,7 +839,7 @@ const writeLodSource = async (options: WriteLodSourceOptions, fs: FileSystem) => await fs.mkdir(dirname(envPathname)); await writeSogSource( - envSource!, + envSource!.meta.camera ? withCamera(envSource!, undefined) : envSource!, pool, { filename: envPathname, bundle: false, iterations, webpEffort, createDevice, logging: 'flat' }, fs @@ -887,7 +895,7 @@ const writeLodSource = async (options: WriteLodSourceOptions, fs: FileSystem) => // already in write order, so pass an identity ordering to skip the // writer's own Morton pass. const unitSource = positionsFromSlim( - permuteSource(mainSource, orderedLocal, { lod: lodValue }), + permuteSource(unitParent, orderedLocal, { lod: lodValue }), slim, orderedIndices ); const identity = new Uint32Array(totalIndices); diff --git a/src/lib/writers/write-sog.ts b/src/lib/writers/write-sog.ts index 4fa38b01..a209e737 100644 --- a/src/lib/writers/write-sog.ts +++ b/src/lib/writers/write-sog.ts @@ -405,7 +405,8 @@ const writeSogSource = async ( scales: { codebook: scalesCodebook, files: ['scales.webp'] }, quats: { files: ['quats.webp'] }, sh0: { codebook: colorsCodebook, files: ['sh0.webp'] }, - ...(shN ? { shN } : {}) + ...(shN ? { shN } : {}), + ...(meta.camera ? { camera: meta.camera } : {}) }; const metaJson = (new TextEncoder()).encode(JSON.stringify(metaObj)); const metaFilename = zipFs ? 'meta.json' : outputFilename; diff --git a/test/cli.test.mjs b/test/cli.test.mjs index 5f989fa1..2f80ef04 100644 --- a/test/cli.test.mjs +++ b/test/cli.test.mjs @@ -12,9 +12,13 @@ const __dirname = dirname(fileURLToPath(import.meta.url)); const rootDir = dirname(__dirname); const cliArgsEnvName = 'SPLAT_TRANSFORM_CLI_TEST_ARGS'; +// From source the WebP codec can't find its wasm next to the module, so point +// it at lib/ (as the in-process tests do) for runs that write SOG. const cliBootstrap = ` const cliArgs = JSON.parse(process.env.${cliArgsEnvName}); process.argv = ['node', 'src/cli/index.ts', ...cliArgs]; +const { WebPCodec } = await import('./src/lib/index.ts'); +WebPCodec.wasmUrl = new URL('./lib/webp.wasm', 'file://' + process.cwd() + '/').href; const { main } = await import('./src/cli/index.ts'); await main(); `; @@ -258,3 +262,45 @@ describe('CLI image sequences', () => { } }); }); + +describe('CLI --camera-from', () => { + it('records --camera-from in SOG and streamed SOG output', async () => { + const { mkdtemp, readFile: readFileFs, rm, writeFile } = await import('node:fs/promises'); + const { tmpdir } = await import('node:os'); + const { join } = await import('node:path'); + const { createTestDataTable, encodePlyBinary } = await import('./helpers/test-utils.mjs'); + const dir = await mkdtemp(join(tmpdir(), 'st-camera-from-cli-')); + try { + const cameraEntry = (id, position) => ({ + id, img_name: `${id}`, width: 800, height: 600, position, + rotation: [[1, 0, 0], [0, 1, 0], [0, 0, 1]], fx: 700, fy: 700 + }); + await writeFile(join(dir, 'in.ply'), encodePlyBinary(createTestDataTable(16))); + await writeFile(join(dir, 'cameras.json'), JSON.stringify([cameraEntry(0, [0, 0, 0]), cameraEntry(1, [1, 2, 3])])); + const expected = { + convention: 'opencv', + rest: { position: [1, 2, 3], rotation: [0, 0, 0, 1] }, + intrinsics: { fx: 700, fy: 700, cx: 400, cy: 300, width: 800, height: 600 } + }; + const readJson = async path => JSON.parse(await readFileFs(path, 'utf-8')); + + for (const out of ['sog/meta.json', 'lod/lod-meta.json']) { + const result = await runCli([ + '--gpu', 'cpu', '-w', '-i', '1', + join(dir, 'in.ply'), '--camera-from', `${join(dir, 'cameras.json')}:1`, + join(dir, out) + ]); + assert.strictEqual(result.code, 0, `CLI failed:\n${result.stderr}\n${result.stdout}`); + assert.deepStrictEqual((await readJson(join(dir, out))).camera, expected, out); + } + + const rejected = await runCli([ + '--gpu', 'cpu', join(dir, 'in.ply'), join(dir, 'x.sog'), '--camera-from', join(dir, 'cameras.json') + ]); + assert.notStrictEqual(rejected.code, 0); + assert.match(rejected.stderr + rejected.stdout, /--camera-from applies to an input file/); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/test/sog-camera.test.mjs b/test/sog-camera.test.mjs new file mode 100644 index 00000000..ca48dc71 --- /dev/null +++ b/test/sog-camera.test.mjs @@ -0,0 +1,207 @@ +/** + * The optional SOG `camera` block: + * + * - built from a 3DGS training cameras.json entry; + * - carried through .sog / meta.json / lod-meta.json read and write unchanged + * (unknown keys included), once at the top of lod-meta.json; + * - moved with the scene by transform actions. + */ + +import assert from 'node:assert'; +import { dirname, join } from 'node:path'; +import { describe, it } from 'node:test'; +import { fileURLToPath } from 'node:url'; + +import { Quat, Vec3 } from 'playcanvas'; + +import { createChunkDataPool } from '../src/lib/chunk/index.js'; +import { dataTableToChunkSource, materializeToDataTable } from '../src/lib/compat/data-table.js'; +import { + Column, DataTable, MemoryFileSystem, MemoryReadFileSystem, Transform, WebPCodec, + processSourceBridged, readFile, sogCameraFromCamerasJson, transformSogCamera, withCamera, writeSource +} from '../src/lib/index.js'; +import { writeLodSource } from '../src/lib/writers/write-lod.js'; + +const __dirname = dirname(fileURLToPath(import.meta.url)); +WebPCodec.wasmUrl = join(__dirname, '..', 'lib', 'webp.wasm'); + +const camera = { + convention: 'opencv', + rig: 'camera', + rest: { position: [0.5, -0.25, 1], rotation: [0, 0, 0, 1] }, + intrinsics: { fx: 1194.667, fy: 1194.667, cx: 1024, cy: 576, width: 2048, height: 1152 }, + stereo: { baseline_m: 0.063 }, + focus: { point: [0, 0, 1.68], subject_m: 2.14, near_m: 0.73, far_m: 66.2 }, + vendor: { anything: [1, 'two', { three: 3 }] } +}; + +// A PLY-space table from explicit positions (identity rotation, small splats). +const makeTable = (points) => { + const n = points.length; + const col = i => new Float32Array(points.map(p => p[i])); + const fill = v => new Float32Array(n).fill(v); + return new DataTable([ + new Column('x', col(0)), new Column('y', col(1)), new Column('z', col(2)), + new Column('rot_0', fill(1)), new Column('rot_1', fill(0)), new Column('rot_2', fill(0)), new Column('rot_3', fill(0)), + new Column('scale_0', fill(-3)), new Column('scale_1', fill(-3)), new Column('scale_2', fill(-3)), + new Column('f_dc_0', fill(0)), new Column('f_dc_1', fill(0)), new Column('f_dc_2', fill(0)), + new Column('opacity', fill(0)) + ], Transform.PLY); +}; + +const points = [[0.5, -0.25, 1], [0.5, -0.25, 2], [-1, 1, 3], [2, 0, 4]]; + +const sourceWithCamera = (cam = camera) => withCamera(dataTableToChunkSource(makeTable(points), 1 << 20), cam); + +// Write `source` and return the MemoryFileSystem results. +const write = async (source, filename, outputFormat) => { + const fs = new MemoryFileSystem(); + await writeSource({ filename, outputFormat, source, pool: createChunkDataPool(), options: { iterations: 1 } }, fs); + return fs.results; +}; + +const readBack = async (results, filename, inputFormat) => { + const rfs = new MemoryReadFileSystem(); + for (const [name, data] of results) rfs.set(name.split('/').pop(), data); + const [source] = await readFile({ filename, inputFormat, fileSystem: rfs }); + return source; +}; + +const metaOf = results => JSON.parse(Buffer.from(results.get('meta.json')).toString()); + +describe('sogCameraFromCamerasJson', () => { + const entry = (rotation, extra = {}) => ({ + id: 0, img_name: '00000', width: 1600, height: 1200, + position: [1, 2, 3], rotation, fx: 1100, fy: 1120, ...extra + }); + + it('maps an INRIA cameras.json entry', () => { + const cam = sogCameraFromCamerasJson([entry([[1, 0, 0], [0, 1, 0], [0, 0, 1]])]); + assert.deepStrictEqual(cam, { + convention: 'opencv', + rest: { position: [1, 2, 3], rotation: [0, 0, 0, 1] }, + intrinsics: { fx: 1100, fy: 1120, cx: 800, cy: 600, width: 1600, height: 1200 } + }); + }); + + it('converts the camera-to-world rotation rows to a quaternion', () => { + // 90 degrees about +y: camera +z (forward) looks down world +x + const cam = sogCameraFromCamerasJson([ + entry([[1, 0, 0], [0, 1, 0], [0, 0, 1]]), + entry([[0, 0, 1], [0, 1, 0], [-1, 0, 0]]) + ], 1); + const [x, y, z, w] = cam.rest.rotation; + const forward = new Quat(x, y, z, w).transformVector(new Vec3(0, 0, 1), new Vec3()); + assert.ok(forward.distance(new Vec3(1, 0, 0)) < 1e-9, `forward ${forward}`); + }); + + it('rejects a missing or malformed entry', () => { + assert.throws(() => sogCameraFromCamerasJson({}), /expected an array/); + assert.throws(() => sogCameraFromCamerasJson([], 0), /no camera at index 0/); + assert.throws(() => sogCameraFromCamerasJson([entry([[1, 0, 0], [0, 1, 0]])]), /3x3 rotation/); + }); +}); + +describe('transformSogCamera', () => { + it('returns an equal copy under identity', () => { + const out = transformSogCamera(camera, new Transform()); + assert.deepStrictEqual(out, camera); + assert.notStrictEqual(out.rest, camera.rest); + }); + + it('moves the rest pose, scales distances, and leaves the rest alone', () => { + const t = new Transform(new Vec3(1, 2, 3), new Quat().setFromEulerAngles(0, 90, 0), 2); + const out = transformSogCamera(camera, t); + + const p = t.transformPoint(new Vec3(0.5, -0.25, 1), new Vec3()); + assert.ok(new Vec3(...out.rest.position).distance(p) < 1e-9); + const q = new Quat(...out.rest.rotation); + const forward = q.transformVector(new Vec3(0, 0, 1), new Vec3()); + assert.ok(forward.distance(new Vec3(1, 0, 0)) < 1e-9, `forward ${forward}`); + + assert.strictEqual(out.stereo.baseline_m, 0.126); + assert.deepStrictEqual(out.focus, { point: [0, 0, 3.36], subject_m: 4.28, near_m: 1.46, far_m: 132.4 }); + assert.deepStrictEqual(out.intrinsics, camera.intrinsics); + assert.deepStrictEqual(out.vendor, camera.vendor); + assert.deepStrictEqual(camera.rest.position, [0.5, -0.25, 1], 'input untouched'); + }); +}); + +describe('SOG camera block', () => { + it('writes the camera to meta.json and reads it back', async () => { + const out = await write(sourceWithCamera(), 'meta.json', 'sog'); + assert.deepStrictEqual(metaOf(out).camera, camera); + + const source = await readBack(out, 'meta.json', 'sog'); + assert.deepStrictEqual(source.meta.camera, camera); + }); + + it('round-trips .sog -> .sog unchanged', async () => { + const first = await write(sourceWithCamera(), 'a.sog', 'sog-bundle'); + const source = await readBack(first, 'a.sog', 'sog'); + const second = await write(source, 'meta.json', 'sog'); + assert.strictEqual(JSON.stringify(metaOf(second).camera), JSON.stringify(camera)); + }); + + it('leaves meta.json unchanged without a camera', async () => { + const out = await write(dataTableToChunkSource(makeTable(points), 1 << 20), 'meta.json', 'sog'); + assert.ok(!('camera' in metaOf(out))); + }); + + it('moves the rest pose with a rotate action', async () => { + // the camera sits on one splat and looks at another 1 unit ahead + const pool = createChunkDataPool(); + const rotated = await processSourceBridged(sourceWithCamera(), [ + { kind: 'rotate', value: new Vec3(0, 90, 0) }, + { kind: 'translate', value: new Vec3(1, 0, 0) } + ], pool); + const out = await write(rotated, 'meta.json', 'sog'); + const cam = metaOf(out).camera; + + // (the writer reorders splats, so match by position) + const table = await materializeToDataTable(await readBack(out, 'meta.json', 'sog'), pool); + const splats = Array.from({ length: table.numRows }, (_, i) => new Vec3(...['x', 'y', 'z'].map(c => table.getColumnByName(c).data[i]))); + const nearest = p => Math.min(...splats.map(s => s.distance(p))); + + const position = new Vec3(...cam.rest.position); + const forward = new Quat(...cam.rest.rotation).transformVector(new Vec3(0, 0, 1), new Vec3()); + assert.ok(nearest(position) < 1e-2, 'camera stays on its splat'); + assert.ok(nearest(position.clone().add(forward)) < 1e-2, 'camera still looks at the next splat'); + }); +}); + +describe('Streamed SOG camera block', () => { + const writeLod = async (source) => { + const fs = new MemoryFileSystem(); + await writeLodSource({ + filename: '/scene/lod-meta.json', + mainSource: source, + envSource: null, + iterations: 1, + chunkCount: 1, + chunkExtent: 16 + }, fs); + return fs.results; + }; + + it('writes the camera once, at the top level of lod-meta.json', async () => { + const out = await writeLod(sourceWithCamera()); + const meta = JSON.parse(Buffer.from(out.get('/scene/lod-meta.json')).toString()); + assert.deepStrictEqual(meta.camera, camera); + assert.deepStrictEqual(Object.keys(meta).slice(0, 3), ['version', 'asset', 'camera']); + + const units = [...out.keys()].filter(k => k.endsWith('/meta.json')); + assert.ok(units.length > 0); + for (const unit of units) { + assert.ok(!('camera' in JSON.parse(Buffer.from(out.get(unit)).toString())), `${unit} repeats the camera`); + } + }); + + it('reads it back from lod-meta.json', async () => { + const out = await writeLod(sourceWithCamera()); + const rfs = new MemoryReadFileSystem(); + for (const [name, data] of out) rfs.set(name.replace('/scene/', ''), data); + const [source] = await readFile({ filename: 'lod-meta.json', inputFormat: 'lod', fileSystem: rfs }); + assert.deepStrictEqual(source.meta.camera, camera); + }); +});