-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmedia.ts
More file actions
327 lines (293 loc) · 11.6 KB
/
Copy pathmedia.ts
File metadata and controls
327 lines (293 loc) · 11.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
/**
* Tier 0 of the ingest pipeline — real measurement, zero tokens.
*
* Everything in this file runs on the actual video file with ffmpeg/ffprobe and
* produces facts, not opinions: duration, resolution, orientation, whether
* there is an audio track, how sharp the picture is, how bright it is, and
* which moments are actually different from one another.
*
* Two of those matter more than they look.
*
* **Sharpness** is measured by running the frames through an edge detector and
* averaging the resulting luma. A clip whose edge energy is near zero is out of
* focus, and no language model needs to be consulted about it. This is what
* lets the pipeline reject unusable footage before spending a single token.
*
* **Scene detection** picks the keyframes. Sampling a video every N seconds
* gives you three near-identical frames of a locked-off shot and misses the one
* moment something happened. Selecting on scene change gives the vision model
* three genuinely different views of the clip, which is both cheaper and
* better — the single highest-leverage decision in the whole token budget.
*
* The binaries ship with the repo via `ffmpeg-static`, so there is nothing to
* install.
*/
import { execFile } from 'node:child_process'
import { promisify } from 'node:util'
import { mkdir, stat } from 'node:fs/promises'
import { dirname } from 'node:path'
import ffmpegPath from 'ffmpeg-static'
import ffprobeStatic from 'ffprobe-static'
const exec = promisify(execFile)
const FFMPEG = (ffmpegPath as unknown as string) ?? 'ffmpeg'
const FFPROBE = ffprobeStatic.path ?? 'ffprobe'
/** ffmpeg is chatty on stderr and returns non-zero for benign reasons; treat output as data. */
async function run(bin: string, args: string[]): Promise<{ stdout: string; stderr: string }> {
try {
return await exec(bin, args, { maxBuffer: 32 * 1024 * 1024, windowsHide: true })
} catch (error) {
const err = error as { stdout?: string; stderr?: string; message?: string }
if (err.stdout != null || err.stderr != null) return { stdout: err.stdout ?? '', stderr: err.stderr ?? '' }
throw error
}
}
export interface VideoProbe {
durationSec: number
width: number
height: number
fps: number
hasAudio: boolean
videoCodec: string
fileSizeMb: number
}
export async function probeVideo(path: string): Promise<VideoProbe> {
const { stdout } = await run(FFPROBE, [
'-v', 'error',
'-print_format', 'json',
'-show_format',
'-show_streams',
path,
])
const data = JSON.parse(stdout) as {
format?: { duration?: string; size?: string }
streams?: {
codec_type?: string
codec_name?: string
width?: number
height?: number
avg_frame_rate?: string
duration?: string
}[]
}
const streams = data.streams ?? []
const video = streams.find((s) => s.codec_type === 'video')
const audio = streams.find((s) => s.codec_type === 'audio')
if (!video) throw new Error(`No video stream in ${path}`)
const [num, den] = (video.avg_frame_rate ?? '0/1').split('/').map(Number)
const fps = den > 0 ? num / den : 0
const durationSec = Number(data.format?.duration ?? video.duration ?? 0)
const bytes = Number(data.format?.size ?? 0)
return {
durationSec: Number(durationSec.toFixed(2)),
width: video.width ?? 0,
height: video.height ?? 0,
fps: Number(fps.toFixed(2)),
hasAudio: Boolean(audio),
videoCodec: video.codec_name ?? 'unknown',
fileSizeMb: Number((bytes / 1_048_576).toFixed(2)),
}
}
/**
* Timestamps where the picture genuinely changes.
*
* Falls back to evenly spaced sampling when a clip is one continuous shot with
* no cuts — which is most raw creator footage, and perfectly normal.
*/
export async function sceneTimestamps(path: string, want: number, durationSec: number): Promise<number[]> {
const { stderr } = await run(FFMPEG, [
'-hide_banner',
'-i', path,
'-vf', "select='gt(scene,0.25)',showinfo",
'-vsync', 'vfr',
'-f', 'null',
'-',
])
const times = [...stderr.matchAll(/pts_time:([0-9.]+)/g)]
.map((match) => Number(match[1]))
.filter((time) => Number.isFinite(time) && time > 0.3 && time < durationSec - 0.2)
// Spread the picks across the clip rather than clustering on one busy second.
const spaced: number[] = []
const minGap = Math.max(0.6, durationSec / (want * 3))
for (const time of times) {
if (spaced.every((existing) => Math.abs(existing - time) > minGap)) spaced.push(time)
if (spaced.length >= want) break
}
if (spaced.length >= want) return spaced.slice(0, want)
// Even sampling for the rest, avoiding the very start and end.
const even = Array.from({ length: want }, (_, index) => ((index + 1) / (want + 1)) * durationSec)
for (const time of even) {
if (spaced.length >= want) break
if (spaced.every((existing) => Math.abs(existing - time) > minGap * 0.5)) spaced.push(time)
}
return spaced.sort((a, b) => a - b).slice(0, want)
}
/** Extract one frame at a timestamp, downscaled — image tokens scale with pixels. */
export async function extractFrame(
source: string,
timestamp: number,
destination: string,
maxWidth = 1024,
): Promise<void> {
await mkdir(dirname(destination), { recursive: true })
await run(FFMPEG, [
'-hide_banner',
'-y',
'-ss', String(timestamp),
'-i', source,
'-frames:v', '1',
'-vf', `scale='min(${maxWidth},iw)':-2:flags=lanczos`,
'-q:v', '3',
destination,
])
}
export interface PictureStats {
/** 0-1. Mean edge energy — a proxy for how sharp the picture is. */
sharpness: number
/** 0-1. Mean luma. */
brightness: number
}
/**
* Measured, not inferred.
*
* `edgedetect` turns the picture into its edges; the average brightness of that
* edge map is high for a crisp image and near zero for a soft one. It is a
* proxy rather than a true focus metric, but it is a real measurement of the
* real file, it costs nothing, and it is more reliable than asking a model to
* judge focus from a compressed still.
*/
export async function pictureStats(path: string): Promise<PictureStats> {
/** Averages the requested signalstats keys across sampled frames. */
const measure = async (filter: string, keys: string[]): Promise<Record<string, number>> => {
const { stdout } = await run(FFPROBE, [
'-v', 'error',
'-f', 'lavfi',
'-i', `movie=${escapeForLavfi(path)},fps=2,${filter}signalstats`,
'-show_entries', `frame_tags=${keys.map((key) => `lavfi.signalstats.${key}`).join(',')}`,
'-of', 'csv=p=0',
])
const rows = stdout
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean)
.map((line) => line.split(','))
const out: Record<string, number> = {}
keys.forEach((key, index) => {
const values = rows.map((row) => Number(row[index])).filter(Number.isFinite)
out[key] = values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : 0
})
return out
}
// Blur the frames again and see how much edge energy survives.
//
// The obvious approach — measure edge energy and call a low number "soft" —
// does not work, and both naive versions were tried against this library
// before this one. Raw edge energy tracks scene brightness and contrast: a
// sunlit pool deck read sixteen times a dim sauna. Dividing by the frame's
// own contrast range fixes that but introduces a worse confound, because it
// then tracks how much fine detail the scene contains — a person against
// smooth water and a tiled steam room both scored *below* a clip that had
// been deliberately blurred.
//
// Comparing a clip against a blurred copy of itself has neither problem. A
// sharp frame loses most of its edges when blurred again; an already-soft one
// barely changes. The result is a ratio of the same scene to itself, so
// brightness, contrast and subject matter all cancel out. Measured here:
//
// deliberately blurred 0.39 <- the only soft clip in the set
// corridor 0.52
// ladle on stones 0.57
// stove / pool deck 0.59
// ice hole / sauna wide 0.64-0.66
// lounge window 0.72
//
// This detects softness, not focus accuracy — it cannot tell a deliberate
// shallow-depth-of-field shot from a missed focus pull, which is exactly why
// it flags for review rather than deciding.
const [sharp, reblurred, base] = await Promise.all([
measure('edgedetect=low=0.1:high=0.3,', ['YAVG']),
measure('gblur=sigma=2,edgedetect=low=0.1:high=0.3,', ['YAVG']),
measure('', ['YAVG']),
])
const edgeLoss = sharp.YAVG > 0 ? 1 - reblurred.YAVG / sharp.YAVG : 0
const FLOOR = 0.35
const SPAN = 0.4
return {
sharpness: Number(Math.max(0, Math.min(1, (edgeLoss - FLOOR) / SPAN)).toFixed(3)),
brightness: Number((base.YAVG / 255).toFixed(3)),
}
}
/**
* Transcode to a small, web-playable clip.
*
* Raw creator footage is hundreds of megabytes. A CRM needs a preview that
* loads instantly in a card, not the master. In production the original would
* sit in object storage and this would be the proxy; here it is what ships in
* the repo, which is why it is capped hard on both resolution and length.
*/
export async function makeWebPreview(
source: string,
destination: string,
options: {
maxSeconds?: number
startAt?: number
crf?: number
/**
* Force a delivery orientation. Creators shoot vertical for social and
* horizontal for everything else, and the brief asks for one or the other —
* so a clip delivered against a vertical requirement is genuinely encoded
* vertical, centre-cropped from the source.
*/
aspect?: 'VERTICAL' | 'HORIZONTAL' | 'SQUARE'
/**
* Gaussian sigma applied before encoding.
*
* Used to give demo clips the defects the scenario says they have. If the
* story is that a creator delivered one soft clip, the file really is soft,
* the pre-filter really measures it, and the reviewer really sees why it
* was rejected. Labelling a perfectly sharp clip "blurry" in a fixture
* would make the whole quality pipeline decorative.
*/
blurSigma?: number
} = {},
): Promise<void> {
const { maxSeconds = 12, startAt = 0, crf = 30, aspect = 'HORIZONTAL', blurSigma = 0 } = options
await mkdir(dirname(destination), { recursive: true })
// 720 on the short edge keeps every clip above the pipeline's own
// low-resolution reject threshold, whichever way up it is.
const target = aspect === 'VERTICAL' ? { w: 720, h: 1280 } : aspect === 'SQUARE' ? { w: 900, h: 900 } : { w: 1280, h: 720 }
// Scale to cover, then centre-crop — no letterboxing, no distortion.
const filter =
`scale=${target.w}:${target.h}:force_original_aspect_ratio=increase:flags=lanczos,` +
`crop=${target.w}:${target.h},setsar=1` +
(blurSigma > 0 ? `,gblur=sigma=${blurSigma.toFixed(2)}` : '')
await run(FFMPEG, [
'-hide_banner',
'-y',
...(startAt > 0 ? ['-ss', String(startAt)] : []),
'-i', source,
'-t', String(maxSeconds),
'-vf', filter,
'-c:v', 'libx264',
'-preset', 'slow',
'-crf', String(crf),
'-pix_fmt', 'yuv420p',
'-c:a', 'aac',
'-b:a', '64k',
'-ac', '1',
'-movflags', '+faststart',
destination,
])
await stat(destination)
}
export function deriveOrientationFromSize(width: number, height: number): 'VERTICAL' | 'HORIZONTAL' | 'SQUARE' {
const ratio = width / height
if (ratio > 1.05) return 'HORIZONTAL'
if (ratio < 0.95) return 'VERTICAL'
return 'SQUARE'
}
/** lavfi treats `:` and `\` as syntax; Windows paths are full of both. */
function escapeForLavfi(path: string): string {
return path.replace(/\\/g, '/').replace(/:/g, '\\\\:')
}
export const FFMPEG_BIN = FFMPEG
export const FFPROBE_BIN = FFPROBE