Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 12 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -285,12 +285,22 @@ a workspace directory containing `scene-manifest.json` with schema
Scene** workflow node to select and validate an existing scene directory.
Scene-capable generators implement `generate_artifact(input_kind,
artifact_path, ...)`; legacy image generators and `POST /generate/from-image`
remain unchanged. The generic `POST /generate/from-artifact` boundary currently
accepts only `scene`, leaving future artifact kinds to separate reviewed changes.
remain unchanged. The generic `POST /generate/from-artifact` boundary accepts
validated `scene` directories and `video` files without converting either to
fake image bytes.
For this first contract, `scene` is model-only and must be declared as the single
`input` value (not inside `inputs`); process and mixed-input scene nodes are rejected.
Model nodes may still accept multiple images and produce a scene.

The dedicated typed-artifact route is used when a model declares exactly
`input: "video"`. Existing video outputs, process video nodes, and `inputs`
arrays containing video remain valid; this feature does not narrow those
extension contracts. Use **Load Video** to import a durable copy under the
workspace before connecting it to a scalar video model input.
The host checks containment, regular-file status, extension, size, and container
signature, while full media decoding remains the extension's responsibility.
Accepted containers are MP4/M4V/MOV, WebM/Matroska, and AVI, up to 8 GiB.


## Modly CLI

Expand Down
2 changes: 1 addition & 1 deletion api/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ uvicorn main:app --host 127.0.0.1 --port 8765 --reload
| GET | `/model/status` | Model download / load status |
| GET | `/model/download` | SSE stream of download progress |
| POST | `/generate/from-image` | Start image-to-3D job |
| POST | `/generate/from-artifact` | Start a typed-artifact model job (`scene` only) |
| POST | `/generate/from-artifact` | Start a typed-artifact model job (`scene` or model-input `video`) |
| GET | `/generate/status/{job_id}` | Poll job status |

## Model
Expand Down
20 changes: 14 additions & 6 deletions api/routers/generation.py
Original file line number Diff line number Diff line change
Expand Up @@ -172,7 +172,8 @@ async def generate_from_artifact(
raise HTTPException(400, str(exc)) from exc

params = {k: v for k, v in request.params.items() if k not in RESERVED_ARTIFACT_PARAMS}
params["scene_manifest_path"] = str(artifact.path)
if artifact.kind == "scene":
params["scene_manifest_path"] = str(artifact.path)
collection = sanitize_collection(request.collection)
job_id = str(uuid.uuid4())
_purge_old_jobs()
Expand Down Expand Up @@ -336,11 +337,18 @@ def progress_cb(pct: int, step: str = "") -> None:
from services.artifact_input import revalidate_artifact_input
model_input = revalidate_artifact_input(registry.WORKSPACE_DIR, model_input)
import inspect
supports_cancel = "cancel_event" in inspect.signature(gen.generate_artifact).parameters
output_path = (
gen.generate_artifact(model_input.kind, model_input.path, params, progress_cb, cancel_event)
if supports_cancel
else gen.generate_artifact(model_input.kind, model_input.path, params, progress_cb)
artifact_parameters = inspect.signature(gen.generate_artifact).parameters
artifact_kwargs = {}
if "cancel_event" in artifact_parameters:
artifact_kwargs["cancel_event"] = cancel_event
if "artifact_snapshot" in artifact_parameters:
artifact_kwargs["artifact_snapshot"] = model_input.snapshot
output_path = gen.generate_artifact(
model_input.kind,
model_input.path,
params,
progress_cb,
**artifact_kwargs,
)
else:
import inspect
Expand Down
23 changes: 18 additions & 5 deletions api/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -169,10 +169,20 @@ def decode_model_input(msg: dict):
if "input" not in msg:
return base64.b64decode(msg["image_b64"])
value = msg["input"]
if not isinstance(value, dict) or set(value) != {"kind", "path"}:
if not isinstance(value, dict):
raise ValueError("Typed artifact input must contain exactly kind and path")
kind = value.get("kind")
expected_fields = {"kind", "path", "snapshot"} if kind == "video" else {"kind", "path"}
if set(value) != expected_fields:
if kind == "video":
raise ValueError("Video artifact input requires a well-formed snapshot")
raise ValueError("Typed artifact input must contain exactly kind and path")
from services.artifact_input import TypedArtifactInput, revalidate_artifact_input
typed = TypedArtifactInput(kind=value.get("kind"), path=Path(value.get("path", "")))
snapshot = None
if kind == "video":
from services.video_input import video_snapshot_from_dict
snapshot = video_snapshot_from_dict(value["snapshot"])
typed = TypedArtifactInput(kind=kind, path=Path(value.get("path", "")), snapshot=snapshot)
return revalidate_artifact_input(WORKSPACE_DIR, typed)


Expand Down Expand Up @@ -250,9 +260,12 @@ def main() -> None:
if not isinstance(params, dict):
raise ValueError("Model params must be an object")
from services.artifact_input import RESERVED_ARTIFACT_PARAMS
params = {key: value for key, value in params.items()
if key not in RESERVED_ARTIFACT_PARAMS}
params["scene_manifest_path"] = str(model_input.path)
params = {
key: value for key, value in params.items()
if key not in RESERVED_ARTIFACT_PARAMS
}
if model_input.kind == "scene":
params["scene_manifest_path"] = str(model_input.path)
if msg.get("outputs_dir"):
gen.outputs_dir = Path(msg["outputs_dir"])
gen.outputs_dir.mkdir(parents=True, exist_ok=True)
Expand Down
4 changes: 2 additions & 2 deletions api/schemas/generation.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,8 +12,8 @@ class JobStatus(BaseModel):


class GenerateFromArtifactRequest(BaseModel):
"""Generic typed-artifact request. Only scene is public in this release."""
input_kind: Literal["scene"]
"""Generic typed-artifact request for model-only scene and video inputs."""
input_kind: Literal["scene", "video"]
input_path: str
model_id: str
collection: str = "Workflows"
Expand Down
13 changes: 11 additions & 2 deletions api/services/artifact_input.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,26 +3,32 @@
from pathlib import Path

from services.scene_input import validate_scene_input
from services.video_input import VideoSnapshot, validate_video_input

SUPPORTED_ARTIFACT_INPUTS = frozenset({"scene"})
SUPPORTED_ARTIFACT_INPUTS = frozenset({"scene", "video"})

# Transport parameters set by the host; callers must not be able to forge them.
RESERVED_ARTIFACT_PARAMS = frozenset({
"artifact_path", "input_kind", "input_path", "scene_path", "scene_manifest_path",
"video_path",
})


@dataclass(frozen=True)
class TypedArtifactInput:
kind: str
path: Path
snapshot: VideoSnapshot | None = None


def validate_artifact_input(workspace: Path, kind: str, input_path: str) -> TypedArtifactInput:
if kind not in SUPPORTED_ARTIFACT_INPUTS:
raise ValueError(f"Unsupported artifact input kind: {kind}")
if kind == "scene":
return TypedArtifactInput(kind="scene", path=validate_scene_input(workspace, input_path))
if kind == "video":
path, snapshot = validate_video_input(workspace, input_path)
return TypedArtifactInput(kind="video", path=path, snapshot=snapshot)
raise ValueError(f"Unsupported artifact input kind: {kind}")


Expand All @@ -31,4 +37,7 @@ def revalidate_artifact_input(workspace: Path, value: TypedArtifactInput) -> Typ
relative = value.path.resolve(strict=True).relative_to(workspace.resolve(strict=True))
except (OSError, ValueError) as exc:
raise ValueError("Artifact input is outside the workspace") from exc
return validate_artifact_input(workspace, value.kind, relative.as_posix())
validated = validate_artifact_input(workspace, value.kind, relative.as_posix())
if value.kind == "video" and value.snapshot is not None and validated.snapshot != value.snapshot:
raise ValueError("Video input changed after it was queued")
return validated
17 changes: 14 additions & 3 deletions api/services/extension_process.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,10 @@
import threading
import uuid
from pathlib import Path
from typing import Callable, Optional
from typing import Callable, Optional, TYPE_CHECKING

if TYPE_CHECKING:
from services.video_input import VideoSnapshot

_RUNNER_PATH = Path(__file__).parent.parent / "runner.py"
_MISSING_MODULE_RE = re.compile(r"No module named ['\"]([^'\"]+)['\"]")
Expand Down Expand Up @@ -404,16 +407,24 @@ def generate_artifact(
params: dict,
progress_cb: Optional[Callable[[int, str], None]] = None,
cancel_event: Optional[threading.Event] = None,
artifact_snapshot: Optional["VideoSnapshot"] = None,
) -> Path:
"""Send a typed artifact envelope to the isolated runner."""
from services.artifact_input import TypedArtifactInput, revalidate_artifact_input
from services.generator_registry import WORKSPACE_DIR
from services.video_input import video_snapshot_to_dict

snapshot_payload = (video_snapshot_to_dict(artifact_snapshot)
if input_kind == "video" else None)
validated = revalidate_artifact_input(
WORKSPACE_DIR, TypedArtifactInput(kind=input_kind, path=artifact_path)
WORKSPACE_DIR,
TypedArtifactInput(kind=input_kind, path=artifact_path, snapshot=artifact_snapshot),
)
input_payload = {"kind": validated.kind, "path": str(validated.path)}
if validated.kind == "video":
input_payload["snapshot"] = snapshot_payload
return self._generate_request(
{"input": {"kind": validated.kind, "path": str(validated.path)}},
{"input": input_payload},
params, progress_cb, cancel_event,
)

Expand Down
145 changes: 145 additions & 0 deletions api/services/video_input.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
"""Secure validation for workspace video model inputs."""
from dataclasses import dataclass
import hashlib
import os
import re
import stat
from pathlib import Path, PurePosixPath, PureWindowsPath

MAX_VIDEO_BYTES = 8 * 1024**3
SUPPORTED_VIDEO_EXTENSIONS = frozenset({".mp4", ".m4v", ".mov", ".webm", ".mkv", ".avi"})
_HEADER_BYTES = 64


@dataclass(frozen=True)
class VideoSnapshot:
size: int
mtime_ns: int
device: int
inode: int
header_sha256: str


_SNAPSHOT_FIELDS = frozenset({"size", "mtime_ns", "device", "inode", "header_sha256"})


def video_snapshot_to_dict(snapshot: VideoSnapshot) -> dict:
"""Serialize a validated snapshot for the isolated runner envelope."""
if not isinstance(snapshot, VideoSnapshot):
raise ValueError("Video snapshot is required")
return {
"size": snapshot.size,
"mtime_ns": snapshot.mtime_ns,
"device": snapshot.device,
"inode": snapshot.inode,
"header_sha256": snapshot.header_sha256,
}


def video_snapshot_from_dict(value: object) -> VideoSnapshot:
"""Strictly reconstruct a snapshot received across the process boundary."""
if not isinstance(value, dict) or set(value) != _SNAPSHOT_FIELDS:
raise ValueError("Video snapshot must contain exactly the expected fields")
numeric = ("size", "mtime_ns", "device", "inode")
if any(isinstance(value[field], bool) or not isinstance(value[field], int)
for field in numeric):
raise ValueError("Video snapshot numeric fields must be integers")
if value["size"] <= 0 or any(value[field] < 0 for field in numeric[1:]):
raise ValueError("Video snapshot contains invalid numeric values")
digest = value["header_sha256"]
if not isinstance(digest, str) or re.fullmatch(r"[0-9a-f]{64}", digest) is None:
raise ValueError("Video snapshot contains an invalid header digest")
return VideoSnapshot(
size=value["size"],
mtime_ns=value["mtime_ns"],
device=value["device"],
inode=value["inode"],
header_sha256=digest,
)


def _safe_relative(value: str) -> Path:
if not isinstance(value, str) or not value or value != value.strip() or "\x00" in value:
raise ValueError("Video path must be a nonempty workspace-relative path")
normalized = value.replace("\\", "/")
if (PurePosixPath(normalized).is_absolute() or PureWindowsPath(normalized).is_absolute()
or re.match(r"^[A-Za-z][A-Za-z0-9+.-]*:", normalized)
or re.search(r"%(?:25|2e|2f|5c|00)", normalized, re.I)
or re.search(r"%(?![0-9a-f]{2})", normalized, re.I)
or any(part in ("", ".", "..") for part in normalized.split("/"))):
raise ValueError("Video path must be a safe workspace-relative path")
return Path(*normalized.split("/"))


def _reject_link_components(path: Path, root: Path) -> None:
try:
relative = path.relative_to(root)
except ValueError as exc:
raise ValueError("Video path escapes the workspace") from exc
current = root
for part in relative.parts:
current = current / part
try:
info = current.lstat()
except OSError as exc:
raise ValueError("Video file is missing or unreadable") from exc
is_reparse = bool(getattr(info, "st_file_attributes", 0)
& getattr(stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0))
if current.is_symlink() or is_reparse:
raise ValueError("Video path must not use symlinks or reparse points")


def _signature_matches(suffix: str, header: bytes) -> bool:
if suffix in {".mp4", ".m4v", ".mov"}:
return len(header) >= 12 and header[4:8] == b"ftyp"
if suffix in {".webm", ".mkv"}:
return header.startswith(b"\x1a\x45\xdf\xa3")
if suffix == ".avi":
return len(header) >= 12 and header[:4] == b"RIFF" and header[8:12] == b"AVI "
return False


def validate_video_input(workspace: Path, video_path: str) -> tuple[Path, VideoSnapshot]:
"""Return a canonical file and stable snapshot without decoding the video."""
root = workspace.resolve(strict=True)
candidate = root / _safe_relative(video_path)
_reject_link_components(candidate, root)
suffix = candidate.suffix.lower()
if suffix not in SUPPORTED_VIDEO_EXTENSIONS:
raise ValueError("Video input uses an unsupported file extension")

flags = os.O_RDONLY | getattr(os, "O_BINARY", 0) | getattr(os, "O_NOFOLLOW", 0)
try:
descriptor = os.open(candidate, flags)
except OSError as exc:
raise ValueError("Video file is missing or unreadable") from exc
try:
info = os.fstat(descriptor)
if not stat.S_ISREG(info.st_mode):
raise ValueError("Video input must be a regular file")
if info.st_size <= 0:
raise ValueError("Video input must not be empty")
if info.st_size > MAX_VIDEO_BYTES:
raise ValueError("Video input exceeds the 8 GiB size limit")
header = os.read(descriptor, _HEADER_BYTES)
finally:
os.close(descriptor)

try:
canonical = candidate.resolve(strict=True)
canonical.relative_to(root)
after = candidate.lstat()
except (OSError, ValueError) as exc:
raise ValueError("Video path escapes the workspace") from exc
if not stat.S_ISREG(after.st_mode) or (after.st_dev, after.st_ino) != (info.st_dev, info.st_ino):
raise ValueError("Video file changed while it was being validated")
if not _signature_matches(suffix, header):
raise ValueError("Video file signature does not match its extension")
snapshot = VideoSnapshot(
size=info.st_size,
mtime_ns=info.st_mtime_ns,
device=info.st_dev,
inode=info.st_ino,
header_sha256=hashlib.sha256(header).hexdigest(),
)
return canonical, snapshot
22 changes: 22 additions & 0 deletions api/tests/test_extension_process.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
from pathlib import Path

from services.extension_process import ExtensionProcess, _venv_python
from services.artifact_input import validate_artifact_input


def _make_proc() -> ExtensionProcess:
Expand Down Expand Up @@ -41,6 +42,27 @@ def test_generate_artifact_sends_typed_scene_without_image_bytes(self) -> None:
self.assertEqual(result, manifest)
self.assertEqual(calls, [({"input": {"kind": "scene", "path": str(manifest.resolve())}}, {"quality": "high"})])

def test_generate_artifact_sends_canonical_video_without_image_bytes(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
workspace = Path(tmp) / "workspace"
video = workspace / "Workflows" / "clip.mp4"
video.parent.mkdir(parents=True)
video.write_bytes(b"\x00\x00\x00\x18ftypisom\x00\x00\x02\x00isomiso2")
proc = _make_proc()
calls = []
proc._generate_request = lambda payload, params, progress, cancel: calls.append(payload) or video
queued = validate_artifact_input(workspace, "video", "Workflows/clip.mp4")
with patch("services.generator_registry.WORKSPACE_DIR", workspace):
proc.generate_artifact("video", video, {}, artifact_snapshot=queued.snapshot)
self.assertEqual(calls[0]["input"]["kind"], "video")
self.assertEqual(calls[0]["input"]["path"], str(video.resolve()))
self.assertEqual(calls[0]["input"]["snapshot"]["size"], len(video.read_bytes()))

def test_generate_artifact_requires_original_video_snapshot(self) -> None:
proc = _make_proc()
with self.assertRaisesRegex(ValueError, "snapshot"):
proc.generate_artifact("video", Path("clip.mp4"), {})

def test_read_loop_writes_sentinel_to_own_queue_only(self) -> None:
proc = _make_proc()

Expand Down
Loading