Compare commits

...
18 Commits
Author SHA1 Message Date
de4dn0te fb651fb147 Merge pull request 'v0.2.0' (#6) from rembg into main
Reviewed-on: #6
2026-07-14 09:42:22 +00:00
de4dn0te cc58bd54d4 Update readme 2026-07-14 09:41:59 +00:00
de4dn0te 7fcbe8d8e7 Cleanup and deterministic output 2026-07-14 09:41:59 +00:00
de4dn0te 3c48ed082b Multiple output formats 2026-07-14 09:41:59 +00:00
de4dn0te bdb69239bc Tiff intermediary 2026-07-14 09:41:59 +00:00
de4dn0te e5601871ac Merge pull request 'Aftermath of CUDA reconfig' (#5) from rembg into main
Reviewed-on: #5
2026-06-29 06:53:09 +00:00
de4dn0te d9ed47aaba Updated Readme 2026-06-29 09:51:27 +03:00
de4dn0te 4a90885527 Cleanup 2026-06-29 09:31:35 +03:00
de4dn0te 931b2af262 EOD 2026-06-26 15:39:39 +03:00
de4dn0te a8f92251a3 Fixed CUDA version mismatch 2026-06-23 14:47:34 +03:00
de4dn0te 1b5f92ccb1 EoD 2026-06-16 15:50:25 +03:00
de4dn0te cba969a113 Change to prores 2026-05-29 15:24:29 +03:00
de4dn0te 855bb4b047 Clean interrupt and cleanup 2026-05-29 13:22:49 +03:00
de4dn0te 9ce219be50 Merge pull request 'Import "Delete .gitmodules" (#4) from main into rembg
Reviewed-on: #4
2026-05-29 07:21:10 +00:00
de4dn0te 355521bac5 Delete .gitmodules 2026-05-29 07:15:39 +00:00
de4dn0te dfa4244e7b Merge pull request 'Goalpost shift' (#3) from rembg into main
Reviewed-on: #3
2026-05-29 07:14:56 +00:00
de4dn0te a9d5f7b97c Automatic cleanup 2026-05-29 10:10:28 +03:00
Uula Ilola (LAB) ceb3db92e6 Another AI hallucination special 2026-05-26 13:11:28 +03:00
16 changed files with 482 additions and 1655 deletions
+9 -3
View File
@@ -1,3 +1,9 @@
/build/** .venv
/lib/yt-dlp /processed
/testfootage/ /frames
/export
*.mp4
*.webm
*.mov
*.mkv
__pycache__
-6
View File
@@ -1,6 +0,0 @@
[submodule "lib/sys"]
path = lib/sys
url = https://go.googlesource.com/sys
[submodule "lib/fsnotify"]
path = lib/fsnotify
url = https://github.com/fsnotify/fsnotify.git
+22
View File
@@ -0,0 +1,22 @@
MIT License
Copyright (c) 2021 Seth Tribbey
Copyright (c) 2026 Uula Ilola
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
Binary file not shown.
Binary file not shown.
Submodule lib/fsnotify deleted from 20b1e15ef3
Submodule lib/sys deleted from 99666ae32e
+6
View File
@@ -0,0 +1,6 @@
{
"venvPath": ".",
"venv": ".venv",
"reportUnusedCallResult": false,
"reportAny": false
}
+57 -36
View File
@@ -1,43 +1,64 @@
# Käkisalmi protoyyppi # rembg_for_video
<hr> Uses [ffmpeg-python](https://github.com/kkroening/ffmpeg-python) and [rembg](https://github.com/danielgatis/rembg) to attempt removal of a background from a video file.
## Yleiset Ideat Based on project [rembg_from_video](https://github.com/seth-tribbey/rembg_from_video) by [seth-tribbey](https://github.com/seth-tribbey)
Jos toteutukseen halutaan ulkoasun puolelta vain taustanpoisto ja henkilön asetus toisen taustan päälle ja *ehkä* esim. tausta vaihtuu välillä tai vierii yms., tämän voi varmaankin hyvin toteuttaa suoraan pelkästään ffmpeg:llä tai muulla samanlaisella. Jos halutaan paljon monimutkaisempia efektejä, voidaan siirtyä täysin takaisin AE (After Effects) pipelinen pariin. Tämän osion toteutus ffmpeg:llä ei tietenkään tarkoita, etteikö kumpaakin pipelinea voisi käyttää. ## Installation:
<hr> Currently i've gotten rembg to work only with Python **3.12** because of onnxruntime's shenanigans X/
```bash
python3.12 -m venv .venv
pip install -r requirements.txt
```
> **Script for fixing the cudnn path on Linux:** <br>
> export LD_LIBRARY_PATH=/path/to/kakisalmi/.venv/lib/python3.12/site-packages/nvidia/cudnn/lib:$LD_LIBRARY_PATH
## Video Pipeline Implementation &nbsp;
> *Tämänhetkiset prototyypit vain suuntaa antavat. Älä kiitos käytä prod ympäristössä. Kaikki versiot sisältävät runsaasti AI:n käsityötä, joten - kuten viittasin - en suosittele oikeaan implementatioon.* ## Usage:
```
python .\rembg_video.py [-h] [--help] [-o] [--model] [--workers] [--smooth] [--smooth-workers] [--buffer-size] [--output-type] input
```
<style>
table {
border-collapse: separate;
border-spacing: 0;
border-radius: 5px;
overflow: hidden;
}
th,
td {
border: 1px solid #a0a0a0;
padding: 8px 10px;
border-radius: 8px;
text-align: center;
}
td:first-child,
th:first-child {
text-align: center;
}
td:nth-child(2),
th:nth-child(2) {
text-align: left;
}
</style>
<div style="display: flex; gap: 20px; align-items: flex-start;">
<table>
<tr><th colspan="2">Optional Arguments:</th></tr>
<tr><td>-o</td><td>Output path </td></tr>
<tr><td>-h --help</td><td>Show help</td></tr>
<tr><td>--model</td><td>Choose model for rembg</td></tr>
<tr><td>--workers</td><td>Number of concurrent process workers</td></tr>
<tr><td>--smooth</td><td>Size of window for temporal smoothing</td></tr>
<tr><td>--smooth-workers</td><td>Number of CPU threads for temporal smoothing</td></tr>
<tr><td>--buffer-size</td><td>Set size of processing buffer</td></tr>
<tr><td>--output-type</td><td>Choose how video is exported
<br>"complete" = Full Color .mov <br>"mask" = Only mask <br>"mask_seq" = Mask image sequence)
</table>
Muuttujat kielen ja kirjaston valitsemiseen on seuraavat: kuinka paljon kokemusta tekijöillä on kieleen, onko kirjasto helppokäyttöinen/riittävän ominaisuusrikas, ja kuinka suorituskykyinen ohjelman halutaan olevan. <table>
<tr><th colspan="2">Positional Arguments:</th></tr>
### Rust with FFmpeg <tr><td>input</td><td>Input Video</td></tr>
</table>
* Teknisesti monimutkaisempi implementaatio kuin Python MoviePy:lla, mutta paljon suorituskyvykkäämpi. </div>
### Go with FFmpeg
* Enimmäkseen mielenkiinnosta mukana, voi olla yhtä hyvä vaihtoehto kuin Rust.
`GOOS=linux GOARCH=amd64 go build -o booth_linux booth_pipeline.go` <br>
`go build -o build/booth_win.exe booth_pipeline.go`
### Python with MoviePy
* Helpompi muokata ( jos vain osaisi pythonia ;) ), mutta varmaankin hitain kaikista, paitsi ehkä Adobe AE pipeline.
## Big Picture- päätökset
Aiotaanko kuvauspisteeseen tehdä minkäänlaista valmistusta e.g. vihreä seinä, valaistus yms? Tarvitaanko AI taustanpoistoa, vai voidaanko hyödyntää oikeita työkaluja, kuten CorridorKey?
Onko tilaajalla jonkinlainen kuva tarkalleen minkälaista toteutusta haluavat? Onko esim. vierivä kuvatausta hyvä?
Minkälainen palvelin museolla on käytössä tällä hetkellä? Voidaanko sitä hyödyntää, vai hankitaanko paikan päälle kone tätä varten? Jos palvelinta voidaan hyödyntää, onko siinä riittävästi resursseja videon käsittelyyn?
### Notes
Nykyinen koodi käyttää hard-coded polkuja koulukoneelta
Koulun systeemin takia myös riippuvuudet ovat local
+346
View File
@@ -0,0 +1,346 @@
import argparse
import io
import os
import ffmpeg
import pathlib
import threading
import numpy as np
from queue import Queue
from shutil import rmtree, move
from PIL import Image
from rembg import new_session, remove
os.environ.setdefault(
"PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True,managed_memory:True"
)
# Parse args
parser = argparse.ArgumentParser(
description="Applies rembg background removal to the frames of a video"
)
parser.add_argument("input", type=str, help="Input video")
parser.add_argument("-o", type=str, default="export", help="Define output path")
parser.add_argument(
"--model",
type=str,
default="u2net_human_seg",
help="rembg model to use (default: u2net_human_seg)",
)
parser.add_argument(
"--workers",
type=int,
default=1,
help="Number of concurrent processing workers (default: 1)",
)
parser.add_argument(
"--smooth",
type=int,
default=3,
help="Temporal mask smoothing window size in frames (default: 3, 0 to disable)",
)
parser.add_argument(
"--buffer-size",
type=int,
default=8,
help="Set maximum number of frames in buffer (default: 8)",
)
parser.add_argument(
"--smooth-workers",
type=int,
default=os.cpu_count() or 4,
help="Number of cpu threads to use for temporal mask smoothing (default: cpu count)",
)
parser.add_argument(
"--output-type",
type=str,
choices=["complete", "mask", "mask_seq"],
default="complete",
help="What way to output keyed video. (default: complete)",
)
args = parser.parse_args()
def is_oom_error(exc):
text = str(exc).lower()
return any(
phrase in text
for phrase in (
"out of memory",
"cuda out of memory",
"failed to allocate",
"oom",
"memory error",
)
)
def image_to_bytes(image):
with io.BytesIO() as buffer:
image.save(buffer, format="TIFF")
return buffer.getvalue()
def remove_with_fallback(image_bytes, session):
try:
rembg_out = remove(image_bytes, session=session)
if args.output_type == "complete":
return rembg_out
elif args.output_type in ("mask", "mask_seq"):
if isinstance(rembg_out, (bytes, bytearray)):
mask_bytes = bytes(rembg_out)
else:
raise TypeError('mask_bytes is not of type "bytes"')
alpha = Image.open(io.BytesIO(mask_bytes)).convert("RGBA").getchannel("A")
return image_to_bytes(alpha)
else:
raise Exception("Unknown output type.")
except Exception as exc:
if is_oom_error(exc):
raise RuntimeError("Insufficient GPU memory!") from exc
else:
raise RuntimeError("Background removal failed unexpectedly.")
# Extract video info
probe = ffmpeg.probe(args.input)
video_stream = next(
(stream for stream in probe["streams"] if stream["codec_type"] == "video"), None
)
if video_stream is None:
raise ValueError(f"No video stream found in input file: {args.input}")
width = int(video_stream["width"])
height = int(video_stream["height"])
whstr = str(width) + "x" + str(height)
framerate = video_stream["avg_frame_rate"]
# Setup workspace folders
frames_dir = os.path.join(str(pathlib.Path(__file__).parent.absolute()), "frames")
processed_dir = os.path.join(str(pathlib.Path(__file__).parent.absolute()), "processed")
smoothed_dir = processed_dir + "_smoothed"
rmtree(frames_dir, ignore_errors=True)
rmtree(processed_dir, ignore_errors=True)
rmtree(smoothed_dir, ignore_errors=True)
# Extract input video frames
os.mkdir(frames_dir)
stream = ffmpeg.input(args.input)
stream = ffmpeg.output(stream, os.path.join(frames_dir, "%04d.tiff"))
ffmpeg.run(stream)
_SENTINEL = object()
try:
# Process frames with pipelined reader -> processors -> writer
if not os.path.isdir(processed_dir):
os.mkdir(processed_dir)
files = sorted(os.listdir(frames_dir))
total_files = len(files)
print(f"Loading rembg session (model={args.model})...", flush=True)
session = new_session(
args.model, providers=["CUDAExecutionProvider", "CPUExecutionProvider"]
)
read_queue = Queue(maxsize=args.buffer_size)
write_queue = Queue(maxsize=args.buffer_size)
errors = []
active_processors = [
args.workers
] # list so processor() can mutate without nonlocal
active_processors_lock = threading.Lock()
def reader():
try:
for idx, file in enumerate(files, 1):
frame_path = os.path.join(frames_dir, file)
with open(frame_path, "rb") as f:
data = f.read()
read_queue.put((idx, file, data))
os.remove(frame_path)
except Exception as e:
errors.append(e)
finally:
# One sentinel per worker so each one knows when to stop
for _ in range(args.workers):
read_queue.put(_SENTINEL)
def processor():
try:
while True:
item = read_queue.get()
if item is _SENTINEL:
break
idx, file, input_data = item
print(f"Processing frame {idx}/{total_files}: {file}", flush=True)
output_data = remove_with_fallback(input_data, session=session)
write_queue.put((idx, file, output_data))
except Exception as e:
errors.append(e)
finally:
# Signal writer only when the last processor finishes
with active_processors_lock:
active_processors[0] -= 1
if active_processors[0] == 0:
write_queue.put(_SENTINEL)
def writer():
try:
while True:
item = write_queue.get()
if item is _SENTINEL:
break
idx, file, output_data = item
with open(os.path.join(processed_dir, file), "wb") as f:
f.write(output_data)
print(f"Written frame {idx}/{total_files}: {file}", flush=True)
except Exception as e:
errors.append(e)
reader_thread = threading.Thread(target=reader, daemon=True)
processor_threads = [
threading.Thread(target=processor, daemon=True) for _ in range(args.workers)
]
writer_thread = threading.Thread(target=writer, daemon=True)
reader_thread.start()
for t in processor_threads:
t.start()
writer_thread.start()
reader_thread.join()
for t in processor_threads:
t.join()
writer_thread.join()
if errors:
raise errors[0]
# Temporal mask smoothing
if args.smooth > 0:
files = sorted(os.listdir(processed_dir))
total = len(files)
window = args.smooth
half = window // 2
print(
f"Applying temporal mask smoothing (window={window}, "
f"workers={args.smooth_workers})...",
flush=True,
)
smoothing_errors = []
progress_lock = threading.Lock()
progress_count = [0]
n_workers = max(1, args.smooth_workers)
# Write smoothed frames to a separate directory rather than
# overwriting processed_dir in place. Overlapping windows mean a
# frame can be a *read* dependency for several write_idx tasks;
# writing in place risked one thread reading a file while another
# was mid-save on it (truncated/corrupt img -> shape errors).
if not os.path.isdir(smoothed_dir):
os.mkdir(smoothed_dir)
def get_alpha(idx):
file = files[idx]
img = Image.open(os.path.join(processed_dir, file)).convert("RGBA")
return np.array(img)[:, :, 3].astype(np.float32)
def smooth_frame(write_idx):
try:
start = max(0, write_idx - half)
end = min(total - 1, write_idx + half)
alphas = np.stack([get_alpha(j) for j in range(start, end + 1)])
smoothed_alpha = np.mean(alphas, axis=0).astype(np.uint8)
filename = files[write_idx]
out_img = Image.open(os.path.join(processed_dir, filename)).convert(
"RGBA"
)
out_arr = np.array(out_img)
out_arr[:, :, 3] = smoothed_alpha
Image.fromarray(out_arr).save(os.path.join(smoothed_dir, filename))
with progress_lock:
progress_count[0] += 1
print(
f"Smoothed frame {progress_count[0]}/{total}: {filename}",
flush=True,
)
except Exception as e:
smoothing_errors.append(e)
smooth_queue = Queue()
for write_idx in range(total):
smooth_queue.put(write_idx)
def smoothing_worker():
while True:
try:
write_idx = smooth_queue.get_nowait()
except Exception:
return
if smoothing_errors:
return
smooth_frame(write_idx)
smoothing_threads = [
threading.Thread(target=smoothing_worker, daemon=True)
for _ in range(n_workers)
]
for t in smoothing_threads:
t.start()
for t in smoothing_threads:
t.join()
if smoothing_errors:
rmtree(smoothed_dir, ignore_errors=True)
raise smoothing_errors[0]
# Swap the smoothed frames in as the new processed_dir contents.
rmtree(processed_dir)
os.rename(smoothed_dir, processed_dir)
if args.output_type != "mask_seq":
# Output video
stream = ffmpeg.input(
os.path.join(processed_dir, "%04d.tiff"),
r=framerate,
f="image2",
s=whstr,
)
if args.output_type == "mask":
output_file = pathlib.Path(args.o) / ("output.mp4")
output_file.parent.mkdir(exist_ok=True, parents=True)
stream = ffmpeg.output(
stream,
str(output_file),
vcodec="libx264",
pix_fmt="gray",
crf=0,
preset="veryslow",
tune="animation",
)
else:
output_file = pathlib.Path(args.o) / ("output.mov")
output_file.parent.mkdir(exist_ok=True, parents=True)
stream = ffmpeg.output(
stream, str(output_file), vcodec="prores_ks", **{"profile:v": "4"}
)
ffmpeg.run(stream, overwrite_output=True)
else:
img_seq_out_folder = os.path.join(args.o, "output_img_seq")
rmtree(img_seq_out_folder, ignore_errors=True)
move(processed_dir, img_seq_out_folder)
except KeyboardInterrupt:
print("\nInterrupted by user")
finally:
print("Removing temporary files...")
rmtree(processed_dir, ignore_errors=True)
rmtree(frames_dir, ignore_errors=True)
rmtree(smoothed_dir, ignore_errors=True)
+42
View File
@@ -0,0 +1,42 @@
attrs==26.1.0
certifi==2026.6.17
charset-normalizer==3.4.7
coloredlogs==15.0.1
ffmpeg-python==0.2.0
flatbuffers==25.12.19
future==1.0.0
humanfriendly==10.0
idna==3.18
ImageIO==2.37.3
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lazy-loader==0.5
llvmlite==0.47.0
mpmath==1.3.0
networkx==3.6.1
numba==0.65.1
numpy==2.4.6
nvidia-cublas-cu12==12.9.2.10
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.10.2.21
nvidia-cufft-cu12==11.4.1.4
nvidia-nvjitlink-cu12==12.9.86
onnxruntime-gpu==1.19.2
packaging==26.2
pillow==12.2.0
platformdirs==4.10.0
pooch==1.9.0
protobuf==7.35.1
PyMatting==1.1.15
referencing==0.37.0
rembg==2.0.76
requests==2.34.2
rpds-py==2026.5.1
scikit-image==0.26.0
scipy==1.18.0
sympy==1.14.0
tifffile==2026.6.1
tqdm==4.68.3
typing_extensions==4.15.0
urllib3==2.7.0
-531
View File
@@ -1,531 +0,0 @@
// Museum Story Booth — Video Processing Pipeline (Go)
// =====================================================
// Produces a panning photo-reel composite from a raw booth recording.
//
// Build
// -----
// go mod init booth
// go get github.com/fsnotify/fsnotify
// go build -o booth booth_pipeline.go
//
// Usage
// -----
// ./booth process recording.mp4 ./output --speaker "Jane Smith"
// ./booth watch ./watch_folder ./output --title "Community Voices"
//
// Dependencies
// ------------
// ffmpeg / ffprobe must be on PATH
// github.com/fsnotify/fsnotify (only needed for 'watch' command)
package main
import (
"encoding/json"
"flag"
"fmt"
"log"
"os"
"os/exec"
"path/filepath"
"sort"
"strconv"
"strings"
"time"
"github.com/fsnotify/fsnotify"
)
// ── Configuration ──────────────────────────────────────────────────────────────
const (
frameInterval = 5 // Extract one key frame every N seconds
panelWidth = 320 // Each filmstrip panel width in px
panelHeight = 240 // Each filmstrip panel height in px
border = 8 // White border around each panel in px
panelGap = 10 // Horizontal gap between panels in px
outputWidth = 1920
outputHeight = 1080
outputFPS = 25
panSpeed = 55 // Pixels/second the reel scrolls
subjectScale = 0.65 // Subject height as fraction of outputHeight
grainStrength = 16
vignetteAngle = "PI/4"
defaultExhibit = "Stories from the Community"
fontBold = `../../lib/fonts/DejaVuSans-Bold.ttf`
fontRegular = `../../lib/fonts/DejaVuSans.ttf`
)
// videoExtensions that trigger the watch pipeline
var videoExtensions = map[string]bool{
".mp4": true, ".mov": true, ".mxf": true, ".avi": true,
}
// ── Metadata ──────────────────────────────────────────────────────────────────
type Metadata struct {
ContentID string `json:"content_id"`
Exhibit string `json:"exhibit"`
Speaker string `json:"speaker"`
DateRecorded string `json:"date_recorded"`
DurationSeconds float64 `json:"duration_seconds"`
PipelineVersion string `json:"pipeline_version"`
}
// ── Helpers ────────────────────────────────────────────────────────────────────
// find ffmpeg and ffprobe
func loadEnv() {
ffmpeg := os.Getenv("FFMPEG_PATH")
if ffmpeg == "" {
ffmpeg = "ffmpeg"
}
ffprobe := os.Getenv("FFPROBE_PATH")
if ffprobe == "" {
ffprobe = "ffprobe"
}
}
// run executes a command, streaming its stderr to our log. Fatal on error.
func run(label string, name string, args ...string) {
if label != "" {
log.Printf(" %s", label)
}
preview := name + " " + strings.Join(args, " ")
if len(preview) > 120 {
preview = preview[:120] + "…"
}
log.Printf(" $ %s", preview)
cmd := exec.Command(name, args...)
cmd.Stderr = os.Stderr
if err := cmd.Run(); err != nil {
log.Fatalf("command failed: %v", err)
}
}
// probeFormat returns the "format" section from ffprobe as a map.
func probeFormat(path string) map[string]interface{} {
out, err := exec.Command(
ffprobe, "-v", "quiet", "-print_format", "json",
"-show_format", "-show_streams", path,
).Output()
if err != nil {
log.Fatalf("ffprobe failed: %v", err)
}
var result map[string]interface{}
if err := json.Unmarshal(out, &result); err != nil {
log.Fatalf("ffprobe JSON parse: %v", err)
}
return result
}
func getDuration(path string) float64 {
info := probeFormat(path)
format := info["format"].(map[string]interface{})
d, _ := strconv.ParseFloat(format["duration"].(string), 64)
return d
}
func getVideoSize(path string) (int, int) {
info := probeFormat(path)
streams := info["streams"].([]interface{})
for _, s := range streams {
stream := s.(map[string]interface{})
if stream["codec_type"] == "video" {
w := int(stream["width"].(float64))
h := int(stream["height"].(float64))
return w, h
}
}
log.Fatalf("no video stream in %s", path)
return 0, 0
}
// escDrawtext escapes characters special to ffmpeg's drawtext filter.
func escDrawtext(s string) string {
s = strings.ReplaceAll(s, `\`, `\\`)
s = strings.ReplaceAll(s, `'`, `\'`)
s = strings.ReplaceAll(s, `:`, `\:`)
return s
}
// glob returns sorted matches, fatal if the pattern errors.
func glob(pattern string) []string {
matches, err := filepath.Glob(pattern)
if err != nil {
log.Fatalf("glob %s: %v", pattern, err)
}
sort.Strings(matches)
return matches
}
// ── Stage 1: Extract key frames ───────────────────────────────────────────────
func extractFrames(inputPath, framesDir string) []string {
log.Println("\n[1/5] Extracting key frames...")
if err := os.MkdirAll(framesDir, 0755); err != nil {
log.Fatalf("mkdir frames: %v", err)
}
vf := fmt.Sprintf(
"fps=1/%d,scale=%d:%d:force_original_aspect_ratio=increase,crop=%d:%d",
frameInterval,
panelWidth, panelHeight,
panelWidth, panelHeight,
)
run("", ffmpeg, "-y", "-i", inputPath,
"-vf", vf, "-q:v", "2",
filepath.Join(framesDir, "frame_%04d.jpg"),
)
frames := glob(filepath.Join(framesDir, "frame_*.jpg"))
log.Printf(" → %d frames extracted", len(frames))
return frames
}
// ── Stage 2: Build filmstrip image ────────────────────────────────────────────
func buildFilmstrip(frames []string, tmpDir string) (stripPath string, stripW, stripH int) {
log.Println("\n[2/5] Building filmstrip image...")
if len(frames) == 0 {
log.Fatal("no frames to stitch — check your input video")
}
n := len(frames)
panelW := panelWidth + border*2
panelH := panelHeight + border*2
stripW = n*panelW + (n-1)*panelGap
stripH = panelH
stripPath = filepath.Join(tmpDir, "filmstrip.png")
// Build the ffmpeg command inputs and filter_complex string
args := []string{"-y"}
for _, f := range frames {
args = append(args, "-i", f)
}
sepia := "colorchannelmixer=" +
"rr=0.393:rg=0.769:rb=0.189:" +
"gr=0.349:gg=0.686:gb=0.168:" +
"br=0.272:bg=0.534:bb=0.131"
var filterParts []string
// Per-frame: scale → white border pad → sepia
for i := range frames {
filterParts = append(filterParts, fmt.Sprintf(
"[%d:v]scale=%d:%d:force_original_aspect_ratio=increase,crop=%d:%d,pad=%d:%d:%d:%d:color=white,%s[p%d]",
i,
panelWidth, panelHeight,
panelWidth, panelHeight,
panelW, panelH, border, border,
sepia, i,
))
}
// Add right-side gap to each panel (except last), then hstack
for i := 0; i < n; i++ {
extra := 0
if i < n-1 {
extra = panelGap
}
filterParts = append(filterParts, fmt.Sprintf(
"[p%d]pad=%d:%d:0:0:color=0x1a1a1a[g%d]",
i, panelW+extra, panelH, i,
))
}
var stackInputs strings.Builder
for i := 0; i < n; i++ {
fmt.Fprintf(&stackInputs, "[g%d]", i)
}
filterParts = append(filterParts,
fmt.Sprintf("%shstack=inputs=%d[strip]", stackInputs.String(), n),
)
args = append(args,
"-filter_complex", strings.Join(filterParts, ";"),
"-map", "[strip]",
"-frames:v", "1",
stripPath,
)
run("Stitching panels with hstack…", ffmpeg, args...)
return stripPath, stripW, stripH
}
// ── Stage 3: Composite ────────────────────────────────────────────────────────
func composite(
inputPath, stripPath string,
stripW, stripH int,
outputPath string,
meta map[string]string,
) {
log.Println("\n[3/5] Compositing panning reel + subject video...")
duration := getDuration(inputPath)
srcW, srcH := getVideoSize(inputPath)
// Subject dimensions, centred
subjH := int(float64(outputHeight) * subjectScale)
subjW := int(float64(subjH) * float64(srcW) / float64(srcH))
subjX := (outputWidth - subjW) / 2
subjY := (outputHeight - subjH) / 2
// Tile copies to cover full pan travel
totalTravel := int(duration*panSpeed) + outputWidth
tileCopies := totalTravel/stripW + 2
if tileCopies < 2 {
tileCopies = 2
}
stripY := (outputHeight - stripH) / 2
panX := fmt.Sprintf("-(t*%d)", panSpeed)
// Drawtext for title card
title := escDrawtext(meta["title"])
date := escDrawtext(meta["date"])
speaker := escDrawtext(meta["speaker"])
drawtext := fmt.Sprintf(
"drawtext=fontfile=%s:text='%s':fontcolor=white:fontsize=28:alpha=0.85:x=(w-text_w)/2:y=h-70:shadowcolor=black:shadowx=1:shadowy=1,"+
"drawtext=fontfile=%s:text='%s':fontcolor=0xdddddd:fontsize=20:alpha=0.7:x=(w-text_w)/2:y=h-36:shadowcolor=black:shadowx=1:shadowy=1",
fontBold, title,
fontRegular, date,
)
if speaker != "" {
drawtext += fmt.Sprintf(
",drawtext=fontfile=%s:text='%s':fontcolor=0xffd580:fontsize=22:alpha=0.90:x=(w-text_w)/2:y=30:shadowcolor=black:shadowx=1:shadowy=1",
fontRegular, speaker,
)
}
filterComplex := strings.Join([]string{
fmt.Sprintf("[1:v]tile=%dx1[strip_tiled]", tileCopies),
fmt.Sprintf("[strip_tiled]crop=%d:%d:'%s':0[reel]", outputWidth, stripH, panX),
fmt.Sprintf("color=c=0x111111:s=%dx%d:r=%d[bg]", outputWidth, outputHeight, outputFPS),
fmt.Sprintf("[bg][reel]overlay=0:%d[bg_reel]", stripY),
fmt.Sprintf("[0:v]scale=%d:%d[subject]", subjW, subjH),
fmt.Sprintf("[bg_reel][subject]overlay=%d:%d[with_subject]", subjX, subjY),
fmt.Sprintf("[with_subject]noise=alls=%d:allf=t+u[grainy]", grainStrength),
fmt.Sprintf("[grainy]vignette=angle=%s:mode=forward[vignetted]", vignetteAngle),
fmt.Sprintf("[vignetted]%s[out]", drawtext),
}, ";")
run("Rendering composite (this may take a while)…",
ffmpeg, "-y",
"-i", inputPath,
"-i", stripPath,
"-filter_complex", filterComplex,
"-map", "[out]",
"-map", "0:a",
"-c:v", "libx264", "-preset", "fast", "-crf", "18",
"-c:a", "aac", "-b:a", "192k",
"-t", strconv.FormatFloat(duration, 'f', 3, 64),
"-r", strconv.Itoa(outputFPS),
outputPath,
)
}
// ── Stage 4: Archive copy ─────────────────────────────────────────────────────
func archiveCopy(displayPath, archivePath string) {
log.Println("\n[4/5] Writing archive copy (ProRes HQ)...")
run("",
ffmpeg, "-y",
"-i", displayPath,
"-c:v", "prores_ks", "-profile:v", "3",
"-c:a", "pcm_s16le",
archivePath,
)
}
// ── Stage 5: Sidecar metadata ─────────────────────────────────────────────────
func writeSidecar(outputPath string, meta map[string]string, duration float64) {
log.Println("\n[5/5] Writing sidecar metadata...")
stem := strings.TrimSuffix(filepath.Base(outputPath), filepath.Ext(outputPath))
exhibit := meta["title"]
if exhibit == "" {
exhibit = defaultExhibit
}
data := Metadata{
ContentID: stem,
Exhibit: exhibit,
Speaker: meta["speaker"],
DateRecorded: meta["date"],
DurationSeconds: math_round(duration, 2),
PipelineVersion: "1.0",
}
sidecarPath := strings.TrimSuffix(outputPath, filepath.Ext(outputPath)) + ".json"
b, _ := json.MarshalIndent(data, "", " ")
if err := os.WriteFile(sidecarPath, b, 0644); err != nil {
log.Fatalf("write sidecar: %v", err)
}
log.Printf(" → %s", sidecarPath)
}
// math_round rounds f to decimalPlaces (avoids importing math just for this).
func math_round(f float64, places int) float64 {
pow := 1.0
for i := 0; i < places; i++ {
pow *= 10
}
return float64(int(f*pow+0.5)) / pow
}
// ── Orchestrator ──────────────────────────────────────────────────────────────
func process(inputPath, outputDir string, meta map[string]string) {
if err := os.MkdirAll(outputDir, 0755); err != nil {
log.Fatalf("mkdir output: %v", err)
}
if meta == nil {
meta = map[string]string{}
}
if meta["title"] == "" {
meta["title"] = defaultExhibit
}
if meta["date"] == "" {
meta["date"] = time.Now().Format("January 02, 2006")
}
stem := strings.TrimSuffix(filepath.Base(inputPath), filepath.Ext(inputPath))
displayOut := filepath.Join(outputDir, stem+"_display.mp4")
archiveOut := filepath.Join(outputDir, stem+"_archive.mov")
log.Printf("\n%s", strings.Repeat("=", 60))
log.Printf(" Processing: %s", filepath.Base(inputPath))
log.Printf("%s", strings.Repeat("=", 60))
// Use a temp dir for intermediate files; cleaned up automatically
tmpDir, err := os.MkdirTemp("", "booth-*")
if err != nil {
log.Fatalf("temp dir: %v", err)
}
defer os.RemoveAll(tmpDir)
frames := extractFrames(inputPath, filepath.Join(tmpDir, "frames"))
stripPath, stripW, stripH := buildFilmstrip(frames, tmpDir)
composite(inputPath, stripPath, stripW, stripH, displayOut, meta)
writeSidecar(displayOut, meta, getDuration(inputPath))
archiveCopy(displayOut, archiveOut)
log.Printf("\n✓ Display copy → %s", displayOut)
log.Printf("✓ Archive copy → %s", archiveOut)
}
// ── Watchdog ──────────────────────────────────────────────────────────────────
func watch(watchDir, outputDir, defaultTitle string) {
watcher, err := fsnotify.NewWatcher()
if err != nil {
log.Fatalf("fsnotify: %v", err)
}
defer watcher.Close()
if err := watcher.Add(watchDir); err != nil {
log.Fatalf("watch dir: %v", err)
}
log.Printf("Watching %q for new recordings… (Ctrl+C to stop)", watchDir)
for {
select {
case event, ok := <-watcher.Events:
if !ok {
return
}
if event.Op&fsnotify.Create == 0 {
continue
}
ext := strings.ToLower(filepath.Ext(event.Name))
if !videoExtensions[ext] {
continue
}
// Wait a moment for the camera software to finish writing
time.Sleep(4 * time.Second)
log.Printf("\n★ New recording detected: %s", filepath.Base(event.Name))
go func(path string) {
process(path, outputDir, map[string]string{
"title": defaultTitle,
"date": time.Now().Format("January 02, 2006"),
})
}(event.Name)
case err, ok := <-watcher.Errors:
if !ok {
return
}
log.Printf("watcher error: %v", err)
}
}
}
// ── CLI ───────────────────────────────────────────────────────────────────────
func main() {
log.SetFlags(0) // cleaner output without timestamps
exe, _ := os.Executable()
envPath := filepath.Join(filepath.Dir(exe), "..", "..", ".env")
if err := godotenv.Load(envPath); err != nil {
log.Fatal("Error loading .env file")
}
loadEnv()
processCmd := flag.NewFlagSet("process", flag.ExitOnError)
pTitle := processCmd.String("title", defaultExhibit, "Exhibit name")
pSpeaker := processCmd.String("speaker", "", "Speaker name (optional)")
watchCmd := flag.NewFlagSet("watch", flag.ExitOnError)
wTitle := watchCmd.String("title", defaultExhibit, "Exhibit name")
if len(os.Args) < 2 {
fmt.Println("Usage:")
fmt.Println(" booth process <input> <output_dir> [--title ...] [--speaker ...]")
fmt.Println(" booth watch <watch_dir> <output_dir> [--title ...]")
os.Exit(1)
}
switch os.Args[1] {
case "process":
processCmd.Parse(os.Args[4:])
if len(os.Args) < 4 {
fmt.Println("Usage: booth process <input> <output_dir>")
os.Exit(1)
}
process(os.Args[2], os.Args[3], map[string]string{
"title": *pTitle,
"speaker": *pSpeaker,
"date": time.Now().Format("January 02, 2006"),
})
case "watch":
watchCmd.Parse(os.Args[4:])
if len(os.Args) < 4 {
fmt.Println("Usage: booth watch <watch_dir> <output_dir>")
os.Exit(1)
}
watch(os.Args[2], os.Args[3], *wTitle)
default:
fmt.Printf("Unknown command %q. Use 'process' or 'watch'.\n", os.Args[1])
os.Exit(1)
}
}
-14
View File
@@ -1,14 +0,0 @@
module booth
go 1.25.0
require github.com/fsnotify/fsnotify v1.7.0
require golang.org/x/sys v0.35.0 // indirect
// Replace directives point the module resolver to your local copies.
// Paths are relative to this go.mod file adjust if your layout differs.
replace (
github.com/fsnotify/fsnotify => ../../lib/fsnotify
golang.org/x/sys => ../../lib/sys
)
-465
View File
@@ -1,465 +0,0 @@
#!/usr/bin/env python3
"""
Museum Story Booth — Video Processing Pipeline (MoviePy)
=========================================================
Produces a panning photo-reel composite from a raw booth recording.
This version uses MoviePy + Pillow for compositing, which means all
the pipeline logic is readable Python rather than FFmpeg filter strings.
Install
-------
pip install moviepy pillow numpy watchdog python-dotenv
MoviePy wraps FFmpeg under the hood for encoding; FFmpeg must be on PATH.
Unlike the pure-FFmpeg version, there is no ImageMagick dependency —
all text rendering uses Pillow directly.
Usage
-----
python booth_moviepy.py process recording.mp4 ./output --speaker "Jane Smith"
python booth_moviepy.py watch ./watch_folder ./output --title "Community Voices"
"""
import argparse
import json
import os
import sys
import time
import tempfile
import subprocess
from datetime import datetime
from pathlib import Path
from dotenv import load_dotenv
import numpy as np
from moviepy import ColorClip, CompositeVideoClip, ImageClip, VideoFileClip
from PIL import Image, ImageDraw, ImageFont
# ── Configuration ──────────────────────────────────────────────────────────────
FRAME_INTERVAL = 5 # Extract one key frame every N seconds
PANEL_WIDTH = 320 # Each filmstrip panel width in px
PANEL_HEIGHT = 240 # Each filmstrip panel height in px
BORDER = 8 # White border around each panel in px
PANEL_GAP = 10 # Horizontal gap between panels in px
OUTPUT_WIDTH = 1920
OUTPUT_HEIGHT = 1080
OUTPUT_FPS = 25
PAN_SPEED = 55 # Pixels/second the reel scrolls
SUBJECT_SCALE = 0.65 # Subject video height as fraction of OUTPUT_HEIGHT
GRAIN_STRENGTH = 16 # Standard deviation of noise (050)
VIGNETTE_POWER = 2.5 # Higher = stronger vignette falloff
EXHIBIT_NAME = "Stories from the Community"
# System fonts — Pillow will fall back to its built-in if these are absent
FONT_PATH_BOLD = "../../lib/fonts/DejaVuSans-Bold.ttf"
FONT_PATH_REGULAR = "../../lib/fonts/DejaVuSans.ttf"
env_path = Path(__file__).parent / ".." / ".." / ".env"
load_dotenv(env_path)
ffmpeg = os.getenv("FFMPEG_PATH", "ffmpeg")
ffprobe = os.getenv("FFPROBE_PATH", "ffprobe")
# ── Font loader ────────────────────────────────────────────────────────────────
def load_font(path: str, size: int) -> ImageFont.FreeTypeFont:
try:
return ImageFont.truetype(path, size)
except (IOError, OSError):
return ImageFont.load_default()
# ── Sepia ──────────────────────────────────────────────────────────────────────
def apply_sepia(img: Image.Image) -> Image.Image:
"""Apply a classic sepia tone to a PIL image."""
arr = np.array(img.convert("RGB"), dtype=np.float32)
r = arr[:, :, 0]
g = arr[:, :, 1]
b = arr[:, :, 2]
arr[:, :, 0] = np.clip(r * 0.393 + g * 0.769 + b * 0.189, 0, 255)
arr[:, :, 1] = np.clip(r * 0.349 + g * 0.686 + b * 0.168, 0, 255)
arr[:, :, 2] = np.clip(r * 0.272 + g * 0.534 + b * 0.131, 0, 255)
return Image.fromarray(arr.astype(np.uint8))
# ── Grain ──────────────────────────────────────────────────────────────────────
def make_grain_effect(strength: int = GRAIN_STRENGTH):
"""
Returns a MoviePy fl_image function that adds temporal film grain.
Using a closure lets us precompute the RNG once and vary per-frame.
"""
rng = np.random.default_rng()
def add_grain(frame: np.ndarray) -> np.ndarray:
noise = rng.normal(0, strength, frame.shape).astype(np.int16)
return np.clip(frame.astype(np.int16) + noise, 0, 255).astype(np.uint8)
return add_grain
# ── Vignette ──────────────────────────────────────────────────────────────────
def make_vignette_mask(width: int, height: int, power: float = VIGNETTE_POWER) -> np.ndarray:
"""
Returns a (H, W, 1) float32 mask where 1.0 = centre (no darkening)
and 0.0 = corners (fully dark). Applied once, reused every frame.
"""
cx, cy = width / 2, height / 2
y, x = np.ogrid[:height, :width]
dist = np.sqrt(((x - cx) / cx) ** 2 + ((y - cy) / cy) ** 2)
mask = np.clip(1.0 - (dist ** power) * 0.6, 0.0, 1.0)
return mask[:, :, np.newaxis] # broadcast over RGB channels
def make_vignette_effect(width: int, height: int):
"""Returns a MoviePy fl_image function that applies a vignette."""
mask = make_vignette_mask(width, height)
def apply_vignette(frame: np.ndarray) -> np.ndarray:
return (frame.astype(np.float32) * mask).clip(0, 255).astype(np.uint8)
return apply_vignette
# ── Stage 1: Extract key frames ───────────────────────────────────────────────
def extract_frames(input_path: Path, frames_dir: Path) -> list[Path]:
"""
Pull frames from the video at FRAME_INTERVAL seconds using MoviePy.
MoviePy gives us direct numpy array access — no FFmpeg subprocess needed.
"""
print("\n[1/5] Extracting key frames...")
frames_dir.mkdir(parents=True, exist_ok=True)
with VideoFileClip(str(input_path)) as clip:
duration = clip.duration
times = list(range(0, int(duration), FRAME_INTERVAL))
out_paths = []
for i, t in enumerate(times):
frame = clip.get_frame(t) # numpy (H, W, 3)
img = Image.fromarray(frame)
img = img.resize(
(PANEL_WIDTH, PANEL_HEIGHT), Image.LANCZOS
)
out_path = frames_dir / f"frame_{i:04d}.jpg"
img.save(out_path, quality=92)
out_paths.append(out_path)
print(f"{len(out_paths)} frames extracted")
return out_paths
# ── Stage 2: Build filmstrip image ────────────────────────────────────────────
def build_filmstrip(frames: list[Path]) -> tuple[Image.Image, int, int]:
"""
Stitch frames into a single wide sepia filmstrip PIL image.
Returns (image, strip_width, strip_height).
Doing this in Pillow is far more readable than the equivalent
FFmpeg hstack filter chain.
"""
print("\n[2/5] Building filmstrip image...")
if not frames:
raise RuntimeError("No frames to stitch — check your input video.")
panel_w = PANEL_WIDTH + BORDER * 2
panel_h = PANEL_HEIGHT + BORDER * 2
n = len(frames)
strip_w = n * panel_w + (n - 1) * PANEL_GAP
strip_h = panel_h
strip = Image.new("RGB", (strip_w, strip_h), color=(26, 26, 26))
for i, frame_path in enumerate(frames):
img = Image.open(frame_path).convert("RGB")
img = img.resize((PANEL_WIDTH, PANEL_HEIGHT), Image.LANCZOS)
img = apply_sepia(img)
# White-bordered panel
panel = Image.new("RGB", (panel_w, panel_h), color=(255, 255, 255))
panel.paste(img, (BORDER, BORDER))
x = i * (panel_w + PANEL_GAP)
strip.paste(panel, (x, 0))
print(f" → Filmstrip: {strip_w} × {strip_h} px ({n} panels)")
return strip, strip_w, strip_h
# ── Stage 3: Composite ────────────────────────────────────────────────────────
def composite(
input_path: Path,
strip_img: Image.Image,
strip_w: int,
strip_h: int,
output_path: Path,
metadata: dict,
) -> None:
"""
The main creative stage — pure MoviePy compositing:
• Dark background
• Filmstrip ImageClip panning behind the speaker
• Speaker video centred and scaled
• Film grain + vignette applied as per-frame functions
• Title card burned in via Pillow (no ImageMagick needed)
"""
print("\n[3/5] Compositing panning reel + subject video...")
clip = VideoFileClip(str(input_path))
duration = clip.duration
# ── Subject clip (centred) ────────────────────────────────────────────────
subj_h = int(OUTPUT_HEIGHT * SUBJECT_SCALE)
subj_w = int(subj_h * clip.w / clip.h)
subj_x = (OUTPUT_WIDTH - subj_w) // 2
subj_y = (OUTPUT_HEIGHT - subj_h) // 2
subject = clip.resized((subj_w, subj_h))
# ── Tiled filmstrip (wide enough to cover full pan travel) ────────────────
total_travel = int(duration * PAN_SPEED) + OUTPUT_WIDTH
tile_copies = max(2, total_travel // strip_w + 2)
tiled_w = tile_copies * strip_w
strip_y = (OUTPUT_HEIGHT - strip_h) // 2
tiled = Image.new("RGB", (tiled_w, strip_h), color=(26, 26, 26))
for i in range(tile_copies):
tiled.paste(strip_img, (i * strip_w, 0))
tiled_np = np.array(tiled)
reel_clip = (
ImageClip(tiled_np)
.with_duration(duration)
# Pan: x moves left over time; y stays fixed at strip_y
.with_position(lambda t: (int(-t * PAN_SPEED), strip_y))
)
# ── Title card (Pillow-rendered, overlaid as ImageClip) ───────────────────
title_img = render_title_card(OUTPUT_WIDTH, OUTPUT_HEIGHT, metadata)
title_clip = (
ImageClip(np.array(title_img))
.with_duration(duration)
.with_opacity(1.0)
.with_position((0, 0))
)
# ── Background ────────────────────────────────────────────────────────────
background = ColorClip(
size=(OUTPUT_WIDTH, OUTPUT_HEIGHT),
color=(17, 17, 17),
duration=duration,
)
# ── Composite layers (bottom → top) ──────────────────────────────────────
composite_clip = CompositeVideoClip([
background,
reel_clip,
subject.with_position((subj_x, subj_y)),
title_clip,
], size=(OUTPUT_WIDTH, OUTPUT_HEIGHT))
# ── Per-frame effects: grain then vignette ────────────────────────────────
grain = make_grain_effect(GRAIN_STRENGTH)
vignette = make_vignette_effect(OUTPUT_WIDTH, OUTPUT_HEIGHT)
final = composite_clip.image_transform(lambda f: vignette(grain(f)))
# ── Render ────────────────────────────────────────────────────────────────
print(" Rendering… (this may take a while)")
final.write_videofile(
str(output_path),
fps=OUTPUT_FPS,
codec="libx264",
audio_codec="aac",
bitrate="8000k",
audio=True,
)
clip.close()
def render_title_card(width: int, height: int, metadata: dict) -> Image.Image:
"""
Render a transparent title card as a PIL RGBA image using Pillow.
Returns an image that can be composited directly — no ImageMagick needed.
"""
img = Image.new("RGBA", (width, height), (0, 0, 0, 0))
draw = ImageDraw.Draw(img)
title = metadata.get("title", EXHIBIT_NAME)
date = metadata.get("date", datetime.now().strftime("%B %d, %Y"))
speaker = metadata.get("speaker", "")
font_title = load_font(FONT_PATH_BOLD, 28)
font_sub = load_font(FONT_PATH_REGULAR, 20)
font_speaker = load_font(FONT_PATH_REGULAR, 22)
def centred_text(text, y, font, color):
bbox = draw.textbbox((0, 0), text, font=font)
tw = bbox[2] - bbox[0]
x = (width - tw) // 2
# Shadow
draw.text((x + 1, y + 1), text, font=font, fill=(0, 0, 0, 160))
draw.text((x, y ), text, font=font, fill=color)
centred_text(title, height - 70, font_title, (255, 255, 255, 217))
centred_text(date, height - 36, font_sub, (221, 221, 221, 178))
if speaker:
centred_text(speaker, 30, font_speaker, (255, 213, 128, 230))
return img
# ── Stage 4: Archive copy ─────────────────────────────────────────────────────
def archive_copy(display_path: Path, archive_path: Path) -> None:
"""Transcode the display MP4 to a lossless ProRes HQ archive."""
print("\n[4/5] Writing archive copy (ProRes HQ)...")
subprocess.run([
ffmpeg, "-y",
"-i", str(display_path),
"-c:v", "prores_ks", "-profile:v", "3",
"-c:a", "pcm_s16le",
str(archive_path),
], check=True, stderr=subprocess.DEVNULL)
# ── Stage 5: Sidecar metadata ─────────────────────────────────────────────────
def write_sidecar(output_path: Path, metadata: dict, duration: float) -> None:
print("\n[5/5] Writing sidecar metadata...")
data = {
"content_id": output_path.stem,
"exhibit": metadata.get("title", EXHIBIT_NAME),
"speaker": metadata.get("speaker", ""),
"date_recorded": metadata.get("date", datetime.now().isoformat()),
"duration_seconds": round(duration, 2),
"pipeline_version": "1.0 (moviepy)",
}
sidecar = output_path.with_suffix(".json")
sidecar.write_text(json.dumps(data, indent=2))
print(f"{sidecar}")
# ── Orchestrator ──────────────────────────────────────────────────────────────
def process(input_path: str, output_dir: str, metadata: dict = None) -> tuple[Path, Path]:
input_path = Path(input_path)
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
if metadata is None:
metadata = {}
metadata.setdefault("title", EXHIBIT_NAME)
metadata.setdefault("date", datetime.now().strftime("%B %d, %Y"))
stem = input_path.stem
display_out = output_dir / f"{stem}_display.mp4"
archive_out = output_dir / f"{stem}_archive.mov"
print(f"\n{'='*60}")
print(f" Processing: {input_path.name}")
print(f"{'='*60}")
with tempfile.TemporaryDirectory() as tmp:
tmp = Path(tmp)
frames = extract_frames(input_path, tmp / "frames")
strip_img, sw, sh = build_filmstrip(frames)
composite(input_path, strip_img, sw, sh, display_out, metadata)
with VideoFileClip(str(input_path)) as c:
duration = c.duration
write_sidecar(display_out, metadata, duration)
archive_copy(display_out, archive_out)
print(f"\n✓ Display copy → {display_out}")
print(f"✓ Archive copy → {archive_out}")
return display_out, archive_out
# ── Watchdog automation ───────────────────────────────────────────────────────
def run_watchdog(watch_dir: str, output_dir: str, default_title: str = EXHIBIT_NAME):
try:
from watchdog.observers import Observer
from watchdog.events import FileSystemEventHandler
except ImportError:
sys.exit("watchdog not installed — run: pip install watchdog")
EXTENSIONS = {".mp4", ".mov", ".mxf", ".avi"}
class BoothHandler(FileSystemEventHandler):
def on_created(self, event):
if event.is_directory:
return
path = Path(event.src_path)
if path.suffix.lower() not in EXTENSIONS:
return
time.sleep(4)
print(f"\n★ New recording detected: {path.name}")
try:
process(str(path), output_dir, {
"title": default_title,
"date": datetime.now().strftime("%B %d, %Y"),
})
except Exception as exc:
print(f"\n✗ Pipeline failed for {path.name}: {exc}")
observer = Observer()
observer.schedule(BoothHandler(), watch_dir, recursive=False)
observer.start()
print(f"Watching {watch_dir!r} for new recordings… (Ctrl+C to stop)")
try:
while True:
time.sleep(1)
except KeyboardInterrupt:
observer.stop()
observer.join()
# ── CLI ───────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(
description="Museum story booth — MoviePy pipeline",
formatter_class=argparse.RawDescriptionHelpFormatter,
)
sub = parser.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("process", help="Process a single recording")
p.add_argument("input")
p.add_argument("output_dir")
p.add_argument("--title", default=EXHIBIT_NAME)
p.add_argument("--speaker", default="")
w = sub.add_parser("watch", help="Watch a folder and auto-process")
w.add_argument("watch_dir")
w.add_argument("output_dir")
w.add_argument("--title", default=EXHIBIT_NAME)
args = parser.parse_args()
if args.cmd == "process":
process(args.input, args.output_dir, {
"title": args.title,
"speaker": args.speaker,
"date": datetime.now().strftime("%B %d, %Y"),
})
elif args.cmd == "watch":
run_watchdog(args.watch_dir, args.output_dir, args.title)
if __name__ == "__main__":
main()
-573
View File
@@ -1,573 +0,0 @@
//! Museum Story Booth — Video Processing Pipeline (Rust)
//! =======================================================
//! Produces a panning photo-reel composite from a raw booth recording.
//!
//! Build
//! -----
//! cargo build --release
//! Cross-compile for Windows from Linux/macOS:
//! cargo build --release --target x86_64-pc-windows-gnu
//!
//! Usage
//! -----
//! booth process recording.mp4 ./output --speaker "Jane Smith"
//! booth watch ./watch_folder ./output --title "Community Voices"
//!
//! Dependencies
//! ------------
//! ffmpeg + ffprobe must be available (configure paths in Config below)
use std::env;
use std::fs;
use std::io::Write;
use std::path::{Path, PathBuf};
use std::process::{Command, Stdio};
use std::thread;
use std::time::Duration;
use notify::{EventKind, RecursiveMode, Watcher};
use serde::{Deserialize, Serialize};
// ── Configuration ─────────────────────────────────────────────────────────────
const FRAME_INTERVAL: u32 = 5; // Extract one key frame every N seconds
const PANEL_WIDTH: u32 = 320;
const PANEL_HEIGHT: u32 = 240;
const BORDER: u32 = 8; // White border around each panel in px
const PANEL_GAP: u32 = 10; // Gap between panels in px
const OUTPUT_WIDTH: u32 = 1920;
const OUTPUT_HEIGHT: u32 = 1080;
const OUTPUT_FPS: u32 = 25;
const PAN_SPEED: u32 = 55; // Pixels/second the reel scrolls
const SUBJECT_SCALE: f64 = 0.65; // Subject height as fraction of OUTPUT_HEIGHT
const GRAIN_STRENGTH: u32 = 16;
const VIGNETTE_ANGLE: &str = "PI/4";
const DEFAULT_EXHIBIT: &str = "Stories from the Community";
const VIDEO_EXTENSIONS: &[&str] = &[".mp4", ".mov", ".mxf", ".avi"];
// ── Runtime config (paths resolved relative to the executable) ────────────────
struct Config {
ffmpeg: PathBuf,
ffprobe: PathBuf,
font_bold: PathBuf,
font_regular: PathBuf,
}
impl Config {
fn new() -> Self {
// Executable lives in build/; repo root is one level up
let exe = env::current_exe().expect("cannot locate executable");
let repo = exe.parent().unwrap().parent().unwrap();
let fonts = repo.join("lib").join("fonts");
// ffmpeg lives at lib/ffmpeg/bin/ in the repo,
// matching the school machine layout where it was installed from source
let ffmpeg_bin = repo.join("lib").join("ffmpeg").join("bin");
Config {
ffmpeg: ffmpeg_bin.join("ffmpeg.exe"),
ffprobe: ffmpeg_bin.join("ffprobe.exe"),
font_bold: fonts.join("DejaVuSans-Bold.ttf"),
font_regular: fonts.join("DejaVuSans.ttf"),
}
}
}
// ── Metadata ──────────────────────────────────────────────────────────────────
#[derive(Serialize, Deserialize)]
struct Sidecar {
content_id: String,
exhibit: String,
speaker: String,
date_recorded: String,
duration_seconds: f64,
pipeline_version: String,
}
// ── Job (what gets processed) ─────────────────────────────────────────────────
#[derive(Clone)]
struct Job {
title: String,
speaker: String,
date: String,
}
impl Job {
fn new(title: &str, speaker: &str) -> Self {
Job {
title: title.to_string(),
speaker: speaker.to_string(),
date: chrono_date(),
}
}
}
fn chrono_date() -> String {
// std only; no chrono crate needed for a simple formatted date
use std::time::{SystemTime, UNIX_EPOCH};
let secs = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap()
.as_secs();
// Days since epoch → approximate date string
let days = secs / 86400;
let (y, m, d) = days_to_ymd(days);
let months = ["January","February","March","April","May","June",
"July","August","September","October","November","December"];
format!("{} {:02}, {}", months[(m - 1) as usize], d, y)
}
/// Naive Gregorian conversion (no external crate needed).
fn days_to_ymd(mut days: u64) -> (u64, u64, u64) {
let mut year = 1970u64;
loop {
let leap = is_leap(year);
let days_in_year = if leap { 366 } else { 365 };
if days < days_in_year { break; }
days -= days_in_year;
year += 1;
}
let leap = is_leap(year);
let month_days: &[u64] = if leap {
&[31,29,31,30,31,30,31,31,30,31,30,31]
} else {
&[31,28,31,30,31,30,31,31,30,31,30,31]
};
let mut month = 1u64;
for &md in month_days {
if days < md { break; }
days -= md;
month += 1;
}
(year, month, days + 1)
}
fn is_leap(y: u64) -> bool { y % 4 == 0 && (y % 100 != 0 || y % 400 == 0) }
// ── Helpers ───────────────────────────────────────────────────────────────────
/// Run a command; stream stderr to our stdout; panic on non-zero exit.
fn run(label: &str, program: &Path, args: &[&str]) {
if !label.is_empty() {
println!(" {label}");
}
let preview = format!("{} {}", program.display(), args.join(" "));
println!(" $ {:.120}", preview);
let status = Command::new(program)
.args(args)
.stderr(Stdio::inherit())
.status()
.unwrap_or_else(|e| panic!("failed to spawn {}: {e}", program.display()));
if !status.success() {
panic!("command failed with status {status}");
}
}
/// Run ffprobe and return its JSON output as a parsed value.
fn probe(ffprobe: &Path, path: &Path) -> serde_json::Value {
let out = Command::new(ffprobe)
.args(["-v", "quiet", "-print_format", "json",
"-show_format", "-show_streams",
path.to_str().unwrap()])
.output()
.expect("ffprobe failed");
serde_json::from_slice(&out.stdout).expect("ffprobe JSON parse error")
}
fn get_duration(ffprobe: &Path, path: &Path) -> f64 {
let v = probe(ffprobe, path);
v["format"]["duration"]
.as_str()
.unwrap()
.parse::<f64>()
.unwrap()
}
fn get_video_size(ffprobe: &Path, path: &Path) -> (u32, u32) {
let v = probe(ffprobe, path);
let streams = v["streams"].as_array().unwrap();
for s in streams {
if s["codec_type"] == "video" {
let w = s["width"].as_u64().unwrap() as u32;
let h = s["height"].as_u64().unwrap() as u32;
return (w, h);
}
}
panic!("no video stream in {}", path.display());
}
/// Escape characters special to FFmpeg's drawtext filter.
fn esc(s: &str) -> String {
s.replace('\\', r"\\")
.replace('\'', r"\'")
.replace(':', r"\:")
}
// ── Stage 1: Extract key frames ───────────────────────────────────────────────
fn extract_frames(cfg: &Config, input: &Path, frames_dir: &Path) -> Vec<PathBuf> {
println!("\n[1/5] Extracting key frames...");
fs::create_dir_all(frames_dir).unwrap();
let vf = format!(
"fps=1/{FRAME_INTERVAL},\
scale={PANEL_WIDTH}:{PANEL_HEIGHT}:force_original_aspect_ratio=increase,\
crop={PANEL_WIDTH}:{PANEL_HEIGHT}"
);
let out_pattern = frames_dir.join("frame_%04d.jpg");
run("", &cfg.ffmpeg, &[
"-y", "-i", input.to_str().unwrap(),
"-vf", &vf,
"-q:v", "2",
out_pattern.to_str().unwrap(),
]);
let mut frames: Vec<PathBuf> = fs::read_dir(frames_dir)
.unwrap()
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| p.extension().map(|x| x == "jpg").unwrap_or(false))
.collect();
frames.sort();
println!("{} frames extracted", frames.len());
frames
}
// ── Stage 2: Build filmstrip ──────────────────────────────────────────────────
fn build_filmstrip(
cfg: &Config,
frames: &[PathBuf],
tmp_dir: &Path,
) -> (PathBuf, u32, u32) {
println!("\n[2/5] Building filmstrip image...");
assert!(!frames.is_empty(), "no frames to stitch");
let n = frames.len() as u32;
let panel_w = PANEL_WIDTH + BORDER * 2;
let panel_h = PANEL_HEIGHT + BORDER * 2;
let strip_w = n * panel_w + (n - 1) * PANEL_GAP;
let strip_h = panel_h;
let out_path = tmp_dir.join("filmstrip.png");
// Build inputs list
let mut args: Vec<String> = vec!["-y".into()];
for f in frames {
args.push("-i".into());
args.push(f.to_str().unwrap().into());
}
let sepia =
"colorchannelmixer=\
rr=0.393:rg=0.769:rb=0.189:\
gr=0.349:gg=0.686:gb=0.168:\
br=0.272:bg=0.534:bb=0.131";
let mut filter_parts: Vec<String> = Vec::new();
// Per-frame: scale → white border pad → sepia
for i in 0..frames.len() {
filter_parts.push(format!(
"[{i}:v]scale={PANEL_WIDTH}:{PANEL_HEIGHT}:\
force_original_aspect_ratio=increase,\
crop={PANEL_WIDTH}:{PANEL_HEIGHT},\
pad={panel_w}:{panel_h}:{BORDER}:{BORDER}:color=white,\
{sepia}[p{i}]"
));
}
// Add right-side gap then hstack
for i in 0..frames.len() {
let extra = if i < frames.len() - 1 { PANEL_GAP } else { 0 };
filter_parts.push(format!(
"[p{i}]pad={}:{panel_h}:0:0:color=0x1a1a1a[g{i}]",
panel_w + extra
));
}
let stack_inputs: String = (0..frames.len()).map(|i| format!("[g{i}]")).collect();
filter_parts.push(format!("{stack_inputs}hstack=inputs={}[strip]", frames.len()));
args.extend([
"-filter_complex".into(), filter_parts.join(";"),
"-map".into(), "[strip]".into(),
"-frames:v".into(), "1".into(),
out_path.to_str().unwrap().into(),
]);
let arg_refs: Vec<&str> = args.iter().map(|s| s.as_str()).collect();
run("Stitching panels with hstack…", &cfg.ffmpeg, &arg_refs);
(out_path, strip_w, strip_h)
}
// ── Stage 3: Composite ────────────────────────────────────────────────────────
fn composite(
cfg: &Config,
input: &Path,
strip_path: &Path,
strip_w: u32,
strip_h: u32,
output: &Path,
job: &Job,
) {
println!("\n[3/5] Compositing panning reel + subject video...");
let duration = get_duration(&cfg.ffprobe, input);
let (src_w, src_h) = get_video_size(&cfg.ffprobe, input);
// Subject dimensions
let subj_h = (OUTPUT_HEIGHT as f64 * SUBJECT_SCALE) as u32;
let subj_w = (subj_h as f64 * src_w as f64 / src_h as f64) as u32;
let subj_x = (OUTPUT_WIDTH - subj_w) / 2;
let subj_y = (OUTPUT_HEIGHT - subj_h) / 2;
// Tile copies
let total_travel = (duration * PAN_SPEED as f64) as u32 + OUTPUT_WIDTH;
let tile_copies = (total_travel / strip_w + 2).max(2);
let strip_y = (OUTPUT_HEIGHT - strip_h) / 2;
let pan_x = format!("-(t*{PAN_SPEED})");
// Drawtext
let title = esc(&job.title);
let date = esc(&job.date);
let speaker = esc(&job.speaker);
let fb = cfg.font_bold.to_str().unwrap();
let fr = cfg.font_regular.to_str().unwrap();
let mut drawtext = format!(
"drawtext=fontfile='{fb}':text='{title}':\
fontcolor=white:fontsize=28:alpha=0.85:\
x=(w-text_w)/2:y=h-70:\
shadowcolor=black:shadowx=1:shadowy=1,\
drawtext=fontfile='{fr}':text='{date}':\
fontcolor=0xdddddd:fontsize=20:alpha=0.7:\
x=(w-text_w)/2:y=h-36:\
shadowcolor=black:shadowx=1:shadowy=1"
);
if !speaker.is_empty() {
drawtext.push_str(&format!(
",drawtext=fontfile='{fr}':text='{speaker}':\
fontcolor=0xffd580:fontsize=22:alpha=0.90:\
x=(w-text_w)/2:y=30:\
shadowcolor=black:shadowx=1:shadowy=1"
));
}
let filter_complex = [
format!("[1:v]tile={tile_copies}x1[strip_tiled]"),
format!("[strip_tiled]crop={OUTPUT_WIDTH}:{strip_h}:'{pan_x}':0[reel]"),
format!("color=c=0x111111:s={OUTPUT_WIDTH}x{OUTPUT_HEIGHT}:r={OUTPUT_FPS}[bg]"),
format!("[bg][reel]overlay=0:{strip_y}[bg_reel]"),
format!("[0:v]scale={subj_w}:{subj_h}[subject]"),
format!("[bg_reel][subject]overlay={subj_x}:{subj_y}[with_subject]"),
format!("[with_subject]noise=alls={GRAIN_STRENGTH}:allf=t+u[grainy]"),
format!("[grainy]vignette=angle={VIGNETTE_ANGLE}:mode=forward[vignetted]"),
format!("[vignetted]{drawtext}[out]"),
].join(";");
let duration_str = format!("{:.3}", duration);
let fps_str = OUTPUT_FPS.to_string();
run("Rendering composite (this may take a while)…", &cfg.ffmpeg, &[
"-y",
"-i", input.to_str().unwrap(),
"-i", strip_path.to_str().unwrap(),
"-filter_complex", &filter_complex,
"-map", "[out]",
"-map", "0:a",
"-c:v", "libx264", "-preset", "fast", "-crf", "18",
"-c:a", "aac", "-b:a", "192k",
"-t", &duration_str,
"-r", &fps_str,
output.to_str().unwrap(),
]);
}
// ── Stage 4: Archive copy ─────────────────────────────────────────────────────
fn archive_copy(cfg: &Config, display: &Path, archive: &Path) {
println!("\n[4/5] Writing archive copy (ProRes HQ)...");
run("", &cfg.ffmpeg, &[
"-y",
"-i", display.to_str().unwrap(),
"-c:v", "prores_ks",
"-profile:v", "3",
"-c:a", "pcm_s16le",
archive.to_str().unwrap(),
]);
}
// ── Stage 5: Sidecar metadata ─────────────────────────────────────────────────
fn write_sidecar(output: &Path, job: &Job, duration: f64) {
println!("\n[5/5] Writing sidecar metadata...");
let stem = output.file_stem().unwrap().to_str().unwrap();
let exhibit = if job.title.is_empty() { DEFAULT_EXHIBIT.into() } else { job.title.clone() };
let data = Sidecar {
content_id: stem.into(),
exhibit,
speaker: job.speaker.clone(),
date_recorded: job.date.clone(),
duration_seconds: (duration * 100.0).round() / 100.0,
pipeline_version: "1.0 (rust)".into(),
};
let sidecar = output.with_extension("json");
let json = serde_json::to_string_pretty(&data).unwrap();
fs::write(&sidecar, json).unwrap();
println!("{}", sidecar.display());
}
// ── Orchestrator ──────────────────────────────────────────────────────────────
fn process(cfg: &Config, input: &Path, output_dir: &Path, job: &Job) {
fs::create_dir_all(output_dir).unwrap();
let stem = input.file_stem().unwrap().to_str().unwrap();
let display_out = output_dir.join(format!("{stem}_display.mp4"));
let archive_out = output_dir.join(format!("{stem}_archive.mov"));
println!("\n{}", "=".repeat(60));
println!(" Processing: {}", input.display());
println!("{}", "=".repeat(60));
let tmp_dir = tempfile::tempdir().expect("failed to create temp dir");
let tmp = tmp_dir.path();
let frames = extract_frames(cfg, input, &tmp.join("frames"));
let (strip_path, strip_w, strip_h) = build_filmstrip(cfg, &frames, tmp);
composite(cfg, input, &strip_path, strip_w, strip_h, &display_out, job);
let duration = get_duration(&cfg.ffprobe, input);
write_sidecar(&display_out, job, duration);
archive_copy(cfg, &display_out, &archive_out);
println!("\n✓ Display copy → {}", display_out.display());
println!("✓ Archive copy → {}", archive_out.display());
}
// ── Watch ─────────────────────────────────────────────────────────────────────
fn watch(cfg: &Config, watch_dir: &Path, output_dir: &Path, default_title: &str) {
use notify::event::CreateKind;
use std::sync::mpsc;
let (tx, rx) = mpsc::channel::<notify::Result<notify::Event>>();
let mut watcher = notify::recommended_watcher(tx)
.expect("failed to create watcher");
watcher.watch(watch_dir, RecursiveMode::NonRecursive)
.expect("failed to watch directory");
println!("Watching {:?} for new recordings… (Ctrl+C to stop)", watch_dir);
for res in rx {
match res {
Ok(event) => {
if !matches!(event.kind, EventKind::Create(CreateKind::File)) {
continue;
}
for path in event.paths {
let ext = path.extension()
.and_then(|e| e.to_str())
.map(|e| format!(".{e}").to_lowercase())
.unwrap_or_default();
if !VIDEO_EXTENSIONS.contains(&ext.as_str()) {
continue;
}
// Wait for camera software to finish writing
thread::sleep(Duration::from_secs(4));
println!("\n★ New recording detected: {}", path.display());
let job = Job::new(default_title, "");
let output_dir = output_dir.to_path_buf();
let cfg_ffmpeg = cfg.ffmpeg.clone();
let cfg_ffprobe = cfg.ffprobe.clone();
let cfg_fb = cfg.font_bold.clone();
let cfg_fr = cfg.font_regular.clone();
thread::spawn(move || {
let cfg = Config {
ffmpeg: cfg_ffmpeg,
ffprobe: cfg_ffprobe,
font_bold: cfg_fb,
font_regular: cfg_fr,
};
process(&cfg, &path, &output_dir, &job);
});
}
}
Err(e) => eprintln!("watcher error: {e}"),
}
}
}
// ── CLI ───────────────────────────────────────────────────────────────────────
fn main() {
let args: Vec<String> = env::args().collect();
let cfg = Config::new();
if args.len() < 2 {
eprintln!("Usage:");
eprintln!(" booth process <input> <output_dir> [--title ...] [--speaker ...]");
eprintln!(" booth watch <watch_dir> <output_dir> [--title ...]");
std::process::exit(1);
}
match args[1].as_str() {
"process" => {
if args.len() < 4 {
eprintln!("Usage: booth process <input> <output_dir>");
std::process::exit(1);
}
let input = PathBuf::from(&args[2]);
let output_dir = PathBuf::from(&args[3]);
let title = flag_value(&args, "--title")
.unwrap_or_else(|| DEFAULT_EXHIBIT.into());
let speaker = flag_value(&args, "--speaker")
.unwrap_or_default();
let job = Job::new(&title, &speaker);
process(&cfg, &input, &output_dir, &job);
}
"watch" => {
if args.len() < 4 {
eprintln!("Usage: booth watch <watch_dir> <output_dir>");
std::process::exit(1);
}
let watch_dir = PathBuf::from(&args[2]);
let output_dir = PathBuf::from(&args[3]);
let title = flag_value(&args, "--title")
.unwrap_or_else(|| DEFAULT_EXHIBIT.into());
watch(&cfg, &watch_dir, &output_dir, &title);
}
cmd => {
eprintln!("Unknown command {cmd:?}. Use 'process' or 'watch'.");
std::process::exit(1);
}
}
}
/// Pull the value after a named flag, e.g. --speaker "Jane" → Some("Jane")
fn flag_value(args: &[String], flag: &str) -> Option<String> {
args.windows(2)
.find(|w| w[0] == flag)
.map(|w| w[1].clone())
}
-25
View File
@@ -1,25 +0,0 @@
[package]
name = "booth-process"
version = "0.1.0"
edition = "2026"
[[bin]]
name = "boothVideoProcess"
path = "src/rust/booth_process.rs"
[dependencies]
# File system event watching (watch mode)
notify = "6.1"
# JSON serialisation for sidecar files + ffprobe parsing
serde = { version = "1", features = ["derive"] }
serde_json = "1"
# Temporary directory (cleaned up automatically on drop)
tempfile = "3"
[profile.release]
# Smaller binary, good for deployment on the museum machine
opt-level = 3
lto = true
strip = true