From 7400f2876a239293884d5ffda5abbbb08f639581 Mon Sep 17 00:00:00 2001 From: Brandon Thomas Date: Sun, 19 Jan 2025 20:12:09 -0500 Subject: [PATCH] wip containerization --- animation/StableAnimator/.dockerignore | 4 ++ animation/StableAnimator/.gitignore | 3 + animation/StableAnimator/Dockerfile | 65 +++++++++++++++++++ .../StableAnimator/command_basic_infer.sh | 35 +++++++--- animation/StableAnimator/extract_skeleton.sh | 16 +++++ 5 files changed, 114 insertions(+), 9 deletions(-) create mode 100644 animation/StableAnimator/.dockerignore create mode 100644 animation/StableAnimator/.gitignore create mode 100644 animation/StableAnimator/Dockerfile mode change 100644 => 100755 animation/StableAnimator/command_basic_infer.sh create mode 100755 animation/StableAnimator/extract_skeleton.sh diff --git a/animation/StableAnimator/.dockerignore b/animation/StableAnimator/.dockerignore new file mode 100644 index 0000000..4ea0141 --- /dev/null +++ b/animation/StableAnimator/.dockerignore @@ -0,0 +1,4 @@ +venv/ +inference/ +Dockerfile +.dockerignore diff --git a/animation/StableAnimator/.gitignore b/animation/StableAnimator/.gitignore new file mode 100644 index 0000000..3b8e555 --- /dev/null +++ b/animation/StableAnimator/.gitignore @@ -0,0 +1,3 @@ +checkpoints/ +inference/ +venv/ diff --git a/animation/StableAnimator/Dockerfile b/animation/StableAnimator/Dockerfile new file mode 100644 index 0000000..1397090 --- /dev/null +++ b/animation/StableAnimator/Dockerfile @@ -0,0 +1,65 @@ +# syntax=docker/dockerfile:1 + +# NB: Does this not work??? +FROM nvidia/cuda:12.0.1-runtime-ubuntu22.04 +#FROM nvidia/cuda:11.8.0-runtime-ubuntu22.04 + +ENV DEBIAN_FRONTEND=noninteractive + +# NB: SadTalker obviously requires ffmpeg +# NB: rsync to monitor time it takes for files to sync in k8s initContainer. +# NB(bt,2023-05-04): Installing lsof, htop, ripgrep, and other debugging tools +# NB: I had to remove `nvidia-driver-515` because it broke CUDA / nvidia-smi +RUN apt-get update && apt-get install -y \ + build-essential \ + curl \ + ffmpeg \ + htop \ + libsndfile1 \ + lsof \ + python3-pip \ + python3.10 \ + python3.10-dev \ + python3.10-venv \ + ripgrep \ + rsync \ + vim \ + wget \ + --no-install-recommends \ + && apt-get clean autoclean && apt-get autoremove -y && rm -rf /var/lib/{apt,dpkg,cache,log}/ + +RUN mkdir /python_install + +WORKDIR /python_install + +COPY requirements.txt . + +# NB: Kubernetes will copy over the venv we create here. +# NB: pip install taken from the StableAnimator README +RUN python3 -m venv python \ + && . python/bin/activate \ + && pip install torch==2.5.1 torchvision==0.20.1 torchaudio==2.5.1 --index-url https://download.pytorch.org/whl/cu124 \ + && pip install torch==2.5.1+cu124 xformers --index-url https://download.pytorch.org/whl/cu124 \ + && pip install -r requirements.txt + +# Non-stable animator dockerfile +#RUN python3 -m venv python \ +# && . python/bin/activate \ +# && pip install torch==1.12.1+cu113 torchvision==0.13.1+cu113 torchaudio==0.12.1 --extra-index-url https://download.pytorch.org/whl/cu113 \ +# && pip install -r requirements.txt + +# Model code is kept separate from the venv +RUN mkdir /model_code + +WORKDIR /model_code + +COPY . . + +RUN apt-get install -y git git-lfs + +RUN pwd +RUN ls -lA + +#RUN cd StableAnimator +RUN git lfs install +RUN git clone https://huggingface.co/FrancisRing/StableAnimator checkpoints diff --git a/animation/StableAnimator/command_basic_infer.sh b/animation/StableAnimator/command_basic_infer.sh old mode 100644 new mode 100755 index 03bf6f2..0387396 --- a/animation/StableAnimator/command_basic_infer.sh +++ b/animation/StableAnimator/command_basic_infer.sh @@ -1,18 +1,35 @@ -CUDA_VISIBLE_DEVICES=0 python inference_basic.py \ - --pretrained_model_name_or_path="path/checkpoints/SVD/stable-video-diffusion-img2vid-xt" \ - --output_dir="path/basic_infer" \ - --validation_control_folder="path/inference/case-1/poses" \ - --validation_image="path/inference/case-1/reference.png" \ +#!/bin/bash + +OUTPUT_DIR="inference/output_1/final_output" + +POSE_FRAME_DIR="inference/output_1/poses" +STARTING_IMAGE="inference/output_1/frames/frame_0.png" + +mkdir -p OUTPUT_DIR + +#CUDA_VISIBLE_DEVICES=0 +python inference_basic.py \ + --pretrained_model_name_or_path="checkpoints/SVD/stable-video-diffusion-img2vid-xt" \ + --output_dir="${OUTPUT_DIR}" \ + --validation_control_folder="${POSE_FRAME_DIR}" \ + --validation_image="${STARTING_IMAGE}" \ --width=576 \ --height=1024 \ --guidance_scale=3.0 \ --num_inference_steps=25 \ - --posenet_model_name_or_path="path/checkpoints/Animation/pose_net.pth" \ - --face_encoder_model_name_or_path="path/checkpoints/Animation/face_encoder.pth" \ - --unet_model_name_or_path="path/checkpoints/Animation/unet.pth" \ + --posenet_model_name_or_path="checkpoints/Animation/pose_net.pth" \ + --face_encoder_model_name_or_path="checkpoints/Animation/face_encoder.pth" \ + --unet_model_name_or_path="checkpoints/Animation/unet.pth" \ --tile_size=16 \ --overlap=4 \ --noise_aug_strength=0.02 \ --frames_overlap=4 \ --decode_chunk_size=4 \ - --gradient_checkpointing \ No newline at end of file + --gradient_checkpointing + + +ffmpeg -framerate 30 \ + -start_number 0 \ + -i "${OUTPUT_DIR}/animated_images/frame_%d.png" \ + -c:v libx264 -pix_fmt yuv420p "${OUTPUT_DIR}/out.mp4" -y + diff --git a/animation/StableAnimator/extract_skeleton.sh b/animation/StableAnimator/extract_skeleton.sh new file mode 100755 index 0000000..7015bb0 --- /dev/null +++ b/animation/StableAnimator/extract_skeleton.sh @@ -0,0 +1,16 @@ +#!/bin/bash + +INPUT_VIDEO="inference/zeihan_trimmed.mp4" + +OUTPUT_FRAMES_DIR="inference/output_1/frames" +OUTPUT_POSE_FRAMES_DIR="inference/output_1/poses" + +mkdir -p $OUTPUT_FRAMES_DIR +mkdir -p $OUTPUT_POSE_FRAMES_DIR + +ffmpeg -i $INPUT_VIDEO -q:v 1 -start_number 0 "${OUTPUT_FRAMES_DIR}/frame_%d.png" + +python DWPose/skeleton_extraction.py \ + --target_image_folder_path="${OUTPUT_FRAMES_DIR}" \ + --ref_image_path="${OUTPUT_FRAMES_DIR}/frame_0.png" \ + --poses_folder_path="${OUTPUT_POSE_FRAMES_DIR}"