mirror of
https://github.com/storytold/storyteller-ml.git
synced 2026-10-09 00:09:55 +00:00
wip containerization
This commit is contained in:
@@ -0,0 +1,4 @@
|
||||
venv/
|
||||
inference/
|
||||
Dockerfile
|
||||
.dockerignore
|
||||
@@ -0,0 +1,3 @@
|
||||
checkpoints/
|
||||
inference/
|
||||
venv/
|
||||
@@ -0,0 +1,65 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
|
||||
# NB: Does this not work???
|
||||
FROM nvidia/cuda:12.0.1-runtime-ubuntu22.04
|
||||
#FROM nvidia/cuda:11.8.0-runtime-ubuntu22.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# NB: SadTalker obviously requires ffmpeg
|
||||
# NB: rsync to monitor time it takes for files to sync in k8s initContainer.
|
||||
# NB(bt,2023-05-04): Installing lsof, htop, ripgrep, and other debugging tools
|
||||
# NB: I had to remove `nvidia-driver-515` because it broke CUDA / nvidia-smi
|
||||
RUN apt-get update && apt-get install -y \
|
||||
build-essential \
|
||||
curl \
|
||||
ffmpeg \
|
||||
htop \
|
||||
libsndfile1 \
|
||||
lsof \
|
||||
python3-pip \
|
||||
python3.10 \
|
||||
python3.10-dev \
|
||||
python3.10-venv \
|
||||
ripgrep \
|
||||
rsync \
|
||||
vim \
|
||||
wget \
|
||||
--no-install-recommends \
|
||||
&& apt-get clean autoclean && apt-get autoremove -y && rm -rf /var/lib/{apt,dpkg,cache,log}/
|
||||
|
||||
RUN mkdir /python_install
|
||||
|
||||
WORKDIR /python_install
|
||||
|
||||
COPY requirements.txt .
|
||||
|
||||
# NB: Kubernetes will copy over the venv we create here.
|
||||
# NB: pip install taken from the StableAnimator README
|
||||
RUN python3 -m venv python \
|
||||
&& . python/bin/activate \
|
||||
&& pip install torch==2.5.1 torchvision==0.20.1 torchaudio==2.5.1 --index-url https://download.pytorch.org/whl/cu124 \
|
||||
&& pip install torch==2.5.1+cu124 xformers --index-url https://download.pytorch.org/whl/cu124 \
|
||||
&& pip install -r requirements.txt
|
||||
|
||||
# Non-stable animator dockerfile
|
||||
#RUN python3 -m venv python \
|
||||
# && . python/bin/activate \
|
||||
# && pip install torch==1.12.1+cu113 torchvision==0.13.1+cu113 torchaudio==0.12.1 --extra-index-url https://download.pytorch.org/whl/cu113 \
|
||||
# && pip install -r requirements.txt
|
||||
|
||||
# Model code is kept separate from the venv
|
||||
RUN mkdir /model_code
|
||||
|
||||
WORKDIR /model_code
|
||||
|
||||
COPY . .
|
||||
|
||||
RUN apt-get install -y git git-lfs
|
||||
|
||||
RUN pwd
|
||||
RUN ls -lA
|
||||
|
||||
#RUN cd StableAnimator
|
||||
RUN git lfs install
|
||||
RUN git clone https://huggingface.co/FrancisRing/StableAnimator checkpoints
|
||||
Regular → Executable
+26
-9
@@ -1,18 +1,35 @@
|
||||
CUDA_VISIBLE_DEVICES=0 python inference_basic.py \
|
||||
--pretrained_model_name_or_path="path/checkpoints/SVD/stable-video-diffusion-img2vid-xt" \
|
||||
--output_dir="path/basic_infer" \
|
||||
--validation_control_folder="path/inference/case-1/poses" \
|
||||
--validation_image="path/inference/case-1/reference.png" \
|
||||
#!/bin/bash
|
||||
|
||||
OUTPUT_DIR="inference/output_1/final_output"
|
||||
|
||||
POSE_FRAME_DIR="inference/output_1/poses"
|
||||
STARTING_IMAGE="inference/output_1/frames/frame_0.png"
|
||||
|
||||
mkdir -p OUTPUT_DIR
|
||||
|
||||
#CUDA_VISIBLE_DEVICES=0
|
||||
python inference_basic.py \
|
||||
--pretrained_model_name_or_path="checkpoints/SVD/stable-video-diffusion-img2vid-xt" \
|
||||
--output_dir="${OUTPUT_DIR}" \
|
||||
--validation_control_folder="${POSE_FRAME_DIR}" \
|
||||
--validation_image="${STARTING_IMAGE}" \
|
||||
--width=576 \
|
||||
--height=1024 \
|
||||
--guidance_scale=3.0 \
|
||||
--num_inference_steps=25 \
|
||||
--posenet_model_name_or_path="path/checkpoints/Animation/pose_net.pth" \
|
||||
--face_encoder_model_name_or_path="path/checkpoints/Animation/face_encoder.pth" \
|
||||
--unet_model_name_or_path="path/checkpoints/Animation/unet.pth" \
|
||||
--posenet_model_name_or_path="checkpoints/Animation/pose_net.pth" \
|
||||
--face_encoder_model_name_or_path="checkpoints/Animation/face_encoder.pth" \
|
||||
--unet_model_name_or_path="checkpoints/Animation/unet.pth" \
|
||||
--tile_size=16 \
|
||||
--overlap=4 \
|
||||
--noise_aug_strength=0.02 \
|
||||
--frames_overlap=4 \
|
||||
--decode_chunk_size=4 \
|
||||
--gradient_checkpointing
|
||||
--gradient_checkpointing
|
||||
|
||||
|
||||
ffmpeg -framerate 30 \
|
||||
-start_number 0 \
|
||||
-i "${OUTPUT_DIR}/animated_images/frame_%d.png" \
|
||||
-c:v libx264 -pix_fmt yuv420p "${OUTPUT_DIR}/out.mp4" -y
|
||||
|
||||
|
||||
Executable
+16
@@ -0,0 +1,16 @@
|
||||
#!/bin/bash
|
||||
|
||||
INPUT_VIDEO="inference/zeihan_trimmed.mp4"
|
||||
|
||||
OUTPUT_FRAMES_DIR="inference/output_1/frames"
|
||||
OUTPUT_POSE_FRAMES_DIR="inference/output_1/poses"
|
||||
|
||||
mkdir -p $OUTPUT_FRAMES_DIR
|
||||
mkdir -p $OUTPUT_POSE_FRAMES_DIR
|
||||
|
||||
ffmpeg -i $INPUT_VIDEO -q:v 1 -start_number 0 "${OUTPUT_FRAMES_DIR}/frame_%d.png"
|
||||
|
||||
python DWPose/skeleton_extraction.py \
|
||||
--target_image_folder_path="${OUTPUT_FRAMES_DIR}" \
|
||||
--ref_image_path="${OUTPUT_FRAMES_DIR}/frame_0.png" \
|
||||
--poses_folder_path="${OUTPUT_POSE_FRAMES_DIR}"
|
||||
Reference in New Issue
Block a user