liveportrait

c36d19db · mashun1 · c36d19db · c36d19db · c36d19db · c36d19db
Commit c36d19db authored Jul 22, 2024 by mashun1
20 changed files
--- a/assets/gradio_title.md
+++ b/assets/gradio_title.md
+<div style="display: flex; justify-content: center; align-items: center; text-align: center;">
+  <div>
+    <h1>LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control</h1>
+    <!-- <span>Add mimics and lip sync to your static portrait driven by a video</span> -->
+    <!-- <span>Efficient Portrait Animation with Stitching and Retargeting Control</span> -->
+    <!-- <br> -->
+    <div style="display: flex; justify-content: center; align-items: center; text-align: center;">
+      <a href="https://arxiv.org/pdf/2407.03168"><img src="https://img.shields.io/badge/arXiv-2407.03168-red"></a>
+      &nbsp;
+      <a href="https://liveportrait.github.io"><img src="https://img.shields.io/badge/Project_Page-LivePortrait-green" alt="Project Page"></a>
+      &nbsp;
+      <a href='https://huggingface.co/spaces/KwaiVGI/liveportrait'><img src='https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-blue'></a>
+      &nbsp;
+      <a href="https://github.com/KwaiVGI/LivePortrait"><img src="https://img.shields.io/badge/Github-Code-blue"></a>
+      &nbsp;
+      <a href="https://github.com/KwaiVGI/LivePortrait"><img src="https://img.shields.io/github/stars/KwaiVGI/LivePortrait
+      "></a>
+    </div>
+  </div>
+</div>
\ No newline at end of file
--- a/icon.png
+++ b/icon.png
--- a/inference.py
+++ b/inference.py
+# coding: utf-8
+import os.path as osp
+import tyro
+import subprocess
+from src.config.argument_config import ArgumentConfig
+from src.config.inference_config import InferenceConfig
+from src.config.crop_config import CropConfig
+from src.live_portrait_pipeline import LivePortraitPipeline
+def partial_fields(target_class, kwargs):
+    return target_class(**{k: v for k, v in kwargs.items() if hasattr(target_class, k)})
+def fast_check_ffmpeg():
+    try:
+        subprocess.run(["ffmpeg", "-version"], capture_output=True, check=True)
+        return True
+    except:
+        return False
+def fast_check_args(args: ArgumentConfig):
+    if not osp.exists(args.source):
+        raise FileNotFoundError(f"source info not found: {args.source}")
+    if not osp.exists(args.driving):
+        raise FileNotFoundError(f"driving info not found: {args.driving}")
+def main():
+    # set tyro theme
+    tyro.extras.set_accent_color("bright_cyan")
+    args = tyro.cli(ArgumentConfig)
+    if not fast_check_ffmpeg():
+        raise ImportError(
+            "FFmpeg is not installed. Please install FFmpeg (including ffmpeg and ffprobe) before running this script. https://ffmpeg.org/download.html"
+        )
+    fast_check_args(args)
+    # specify configs for inference
+    inference_cfg = partial_fields(InferenceConfig, args.__dict__)
+    crop_cfg = partial_fields(CropConfig, args.__dict__)
+    live_portrait_pipeline = LivePortraitPipeline(
+        inference_cfg=inference_cfg,
+        crop_cfg=crop_cfg
+    )
+    # run
+    live_portrait_pipeline.execute(args)
+if __name__ == "__main__":
+    main()
\ No newline at end of file
--- a/model.properties
+++ b/model.properties
+# 模型唯一标识
+modelCode=7815
+# 模型名称
+modelName=liveportrait_pytorch
+# 模型描述
+modelDescription=liveportrait可以实现静态图模仿动态图面部动作。
+# 应用场景
+appScenario=推理,AIGC,零售,广媒,教育
+# 框架类型
+frameType=pytorch
--- a/pretrained_weights/.gitkeep
+++ b/pretrained_weights/.gitkeep
--- a/readme_imgs/alg.png
+++ b/readme_imgs/alg.png
--- a/readme_imgs/d0.gif
+++ b/readme_imgs/d0.gif
--- a/readme_imgs/model.png
+++ b/readme_imgs/model.png
--- a/readme_imgs/s20d19.gif
+++ b/readme_imgs/s20d19.gif
--- a/readme_imgs/s7.jpg
+++ b/readme_imgs/s7.jpg
--- a/readme_imgs/s7d0.gif
+++ b/readme_imgs/s7d0.gif
--- a/readme_official.md
+++ b/readme_official.md
+<h1 align="center">LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control</h1>
+<div align='center'>
+    <a href='https://github.com/cleardusk' target='_blank'><strong>Jianzhu Guo</strong></a><sup> 1†</sup>&emsp;
+    <a href='https://github.com/KwaiVGI' target='_blank'><strong>Dingyun Zhang</strong></a><sup> 1,2</sup>&emsp;
+    <a href='https://github.com/KwaiVGI' target='_blank'><strong>Xiaoqiang Liu</strong></a><sup> 1</sup>&emsp;
+    <a href='https://scholar.google.com/citations?user=t88nyvsAAAAJ&hl' target='_blank'><strong>Zhizhou Zhong</strong></a><sup> 1,3</sup>&emsp;
+    <a href='https://scholar.google.com.hk/citations?user=_8k1ubAAAAAJ' target='_blank'><strong>Yuan Zhang</strong></a><sup> 1</sup>&emsp;
+</div>
+<div align='center'>
+    <a href='https://scholar.google.com/citations?user=P6MraaYAAAAJ' target='_blank'><strong>Pengfei Wan</strong></a><sup> 1</sup>&emsp;
+    <a href='https://openreview.net/profile?id=~Di_ZHANG3' target='_blank'><strong>Di Zhang</strong></a><sup> 1</sup>&emsp;
+</div>
+<div align='center'>
+    <sup>1 </sup>Kuaishou Technology&emsp; <sup>2 </sup>University of Science and Technology of China&emsp; <sup>3 </sup>Fudan University&emsp;
+</div>
+<div align='center'>
+    <small><sup>†</sup> Corresponding author</small>
+</div>
+<br>
+<div align="center">
+  <!-- <a href='LICENSE'><img src='https://img.shields.io/badge/license-MIT-yellow'></a> -->
+  <a href='https://arxiv.org/pdf/2407.03168'><img src='https://img.shields.io/badge/arXiv-LivePortrait-red'></a>
+  <a href='https://liveportrait.github.io'><img src='https://img.shields.io/badge/Project-LivePortrait-green'></a>
+  <a href='https://huggingface.co/spaces/KwaiVGI/liveportrait'><img src='https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-blue'></a>
+  <a href="https://github.com/KwaiVGI/LivePortrait"><img src="https://img.shields.io/github/stars/KwaiVGI/LivePortrait"></a>
+</div>
+<br>
+<p align="center">
+  <img src="./assets/docs/showcase2.gif" alt="showcase">
+  <br>
+  🔥 For more results, visit our <a href="https://liveportrait.github.io/"><strong>homepage</strong></a> 🔥
+</p>
+## 🔥 Updates
+- **`2024/07/17`**: 🍎 We support macOS with Apple Silicon, modified from [jeethu](https://github.com/jeethu)'s PR [#143](https://github.com/KwaiVGI/LivePortrait/pull/143).
+- **`2024/07/10`**: 💪 We support audio and video concatenating, driving video auto-cropping, and template making to protect privacy. More to see [here](assets/docs/changelog/2024-07-10.md).
+- **`2024/07/09`**: 🤗 We released the [HuggingFace Space](https://huggingface.co/spaces/KwaiVGI/liveportrait), thanks to the HF team and [Gradio](https://github.com/gradio-app/gradio)!
+- **`2024/07/04`**: 😊 We released the initial version of the inference code and models. Continuous updates, stay tuned!
+- **`2024/07/04`**: 🔥 We released the [homepage](https://liveportrait.github.io) and technical report on [arXiv](https://arxiv.org/pdf/2407.03168).
+## Introduction
+This repo, named **LivePortrait**, contains the official PyTorch implementation of our paper [LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control](https://arxiv.org/pdf/2407.03168).
+We are actively updating and improving this repository. If you find any bugs or have suggestions, welcome to raise issues or submit pull requests (PR) 💖.
+## 🔥 Getting Started
+### 1. Clone the code and prepare the environment
+```bash
+git clone https://github.com/KwaiVGI/LivePortrait
+cd LivePortrait
+# create env using conda
+conda create -n LivePortrait python==3.9
+conda activate LivePortrait
+# install dependencies with pip (for Linux and Windows)
+pip install -r requirements.txt
+# for macOS with Apple Silicon
+pip install -r requirements_macOS.txt
+```
+**Note:** make sure your system has [FFmpeg](https://ffmpeg.org/download.html) installed, including both `ffmpeg` and `ffprobe`!
+### 2. Download pretrained weights
+The easiest way to download the pretrained weights is from HuggingFace:
+```bash
+# first, ensure git-lfs is installed, see: https://docs.github.com/en/repositories/working-with-files/managing-large-files/installing-git-large-file-storage
+git lfs install
+# clone and move the weights
+git clone https://huggingface.co/KwaiVGI/LivePortrait temp_pretrained_weights
+mv temp_pretrained_weights/* pretrained_weights/
+rm -rf temp_pretrained_weights
+```
+Alternatively, you can download all pretrained weights from [Google Drive](https://drive.google.com/drive/folders/1UtKgzKjFAOmZkhNK-OYT0caJ_w2XAnib) or [Baidu Yun](https://pan.baidu.com/s/1MGctWmNla_vZxDbEp2Dtzw?pwd=z5cn). Unzip and place them in `./pretrained_weights`.
+Ensuring the directory structure is as follows, or contains:
+```text
+pretrained_weights
+├── insightface
+│   └── models
+│       └── buffalo_l
+│           ├── 2d106det.onnx
+│           └── det_10g.onnx
+└── liveportrait
+    ├── base_models
+    │   ├── appearance_feature_extractor.pth
+    │   ├── motion_extractor.pth
+    │   ├── spade_generator.pth
+    │   └── warping_module.pth
+    ├── landmark.onnx
+    └── retargeting_models
+        └── stitching_retargeting_module.pth
+```
+### 3. Inference 🚀
+#### Fast hands-on
+```bash
+# For Linux and Windows
+python inference.py
+# For macOS with Apple Silicon, Intel not supported, this maybe 20x slower than RTX 4090
+PYTORCH_ENABLE_MPS_FALLBACK=1 python inference.py
+```
+If the script runs successfully, you will get an output mp4 file named `animations/s6--d0_concat.mp4`. This file includes the following results: driving video, input image, and generated result.
+<p align="center">
+  <img src="./assets/docs/inference.gif" alt="image">
+</p>
+Or, you can change the input by specifying the `-s` and `-d` arguments:
+```bash
+python inference.py -s assets/examples/source/s9.jpg -d assets/examples/driving/d0.mp4
+# disable pasting back to run faster
+python inference.py -s assets/examples/source/s9.jpg -d assets/examples/driving/d0.mp4 --no_flag_pasteback
+# more options to see
+python inference.py -h
+```
+#### Driving video auto-cropping
+📕 To use your own driving video, we **recommend**:
+ - Crop it to a **1:1** aspect ratio (e.g., 512x512 or 256x256 pixels), or enable auto-cropping by `--flag_crop_driving_video`.
+ - Focus on the head area, similar to the example videos.
+ - Minimize shoulder movement.
+ - Make sure the first frame of driving video is a frontal face with **neutral expression**.
+Below is a auto-cropping case by `--flag_crop_driving_video`:
+```bash
+python inference.py -s assets/examples/source/s9.jpg -d assets/examples/driving/d13.mp4 --flag_crop_driving_video
+```
+If you find the results of auto-cropping is not well, you can modify the `--scale_crop_video`, `--vy_ratio_crop_video` options to adjust the scale and offset, or do it manually.
+#### Motion template making
+You can also use the auto-generated motion template files ending with `.pkl` to speed up inference, and **protect privacy**, such as:
+```bash
+python inference.py -s assets/examples/source/s9.jpg -d assets/examples/driving/d5.pkl
+```
+**Discover more interesting results on our [Homepage](https://liveportrait.github.io)** 😊
+### 4. Gradio interface 🤗
+We also provide a Gradio <a href='https://github.com/gradio-app/gradio'><img src='https://img.shields.io/github/stars/gradio-app/gradio'></a> interface for a better experience, just run by:
+```bash
+# For Linux and Windows:
+python app.py
+# For macOS with Apple Silicon, Intel not supported, this maybe 20x slower than RTX 4090
+PYTORCH_ENABLE_MPS_FALLBACK=1 python app.py
+```
+You can specify the `--server_port`, `--share`, `--server_name` arguments to satisfy your needs!
+🚀 We also provide an acceleration option `--flag_do_torch_compile`. The first-time inference triggers an optimization process (about one minute), making subsequent inferences 20-30% faster. Performance gains may vary with different CUDA versions.
+```bash
+# enable torch.compile for faster inference
+python app.py --flag_do_torch_compile
+```
+**Note**: This method is not supported on Windows and macOS.
+**Or, try it out effortlessly on [HuggingFace](https://huggingface.co/spaces/KwaiVGI/LivePortrait) 🤗**
+### 5. Inference speed evaluation 🚀🚀🚀
+We have also provided a script to evaluate the inference speed of each module:
+```bash
+# For NVIDIA GPU
+python speed.py
+```
+Below are the results of inferring one frame on an RTX 4090 GPU using the native PyTorch framework with `torch.compile`:
+| Model                             | Parameters(M) | Model Size(MB) | Inference(ms) |
+|-----------------------------------|:-------------:|:--------------:|:-------------:|
+| Appearance Feature Extractor      |     0.84      |       3.3      |     0.82      |
+| Motion Extractor                  |     28.12     |       108      |     0.84      |
+| Spade Generator                   |     55.37     |       212      |     7.59      |
+| Warping Module                    |     45.53     |       174      |     5.21      |
+| Stitching and Retargeting Modules |     0.23      |       2.3      |     0.31      |
+*Note: The values for the Stitching and Retargeting Modules represent the combined parameter counts and total inference time of three sequential MLP networks.*
+## Community Resources 🤗
+Discover the invaluable resources contributed by our community to enhance your LivePortrait experience:
+- [ComfyUI-LivePortraitKJ](https://github.com/kijai/ComfyUI-LivePortraitKJ) by [@kijai](https://github.com/kijai)
+- [comfyui-liveportrait](https://github.com/shadowcz007/comfyui-liveportrait) by [@shadowcz007](https://github.com/shadowcz007)
+- [LivePortrait In ComfyUI](https://www.youtube.com/watch?v=aFcS31OWMjE) by [@Benji](https://www.youtube.com/@TheFutureThinker)
+- [LivePortrait hands-on tutorial](https://www.youtube.com/watch?v=uyjSTAOY7yI) by [@AI Search](https://www.youtube.com/@theAIsearch)
+- [ComfyUI tutorial](https://www.youtube.com/watch?v=8-IcDDmiUMM) by [@Sebastian Kamph](https://www.youtube.com/@sebastiankamph)
+- [Replicate Playground](https://replicate.com/fofr/live-portrait) and [cog-comfyui](https://github.com/fofr/cog-comfyui) by [@fofr](https://github.com/fofr)
+And many more amazing contributions from our community!
+## Acknowledgements
+We would like to thank the contributors of [FOMM](https://github.com/AliaksandrSiarohin/first-order-model), [Open Facevid2vid](https://github.com/zhanglonghao1992/One-Shot_Free-View_Neural_Talking_Head_Synthesis), [SPADE](https://github.com/NVlabs/SPADE), [InsightFace](https://github.com/deepinsight/insightface) repositories, for their open research and contributions.
+## Citation 💖
+If you find LivePortrait useful for your research, welcome to 🌟 this repo and cite our work using the following BibTeX:
+```bibtex
+@article{guo2024liveportrait,
+  title   = {LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control},
+  author  = {Guo, Jianzhu and Zhang, Dingyun and Liu, Xiaoqiang and Zhong, Zhizhou and Zhang, Yuan and Wan, Pengfei and Zhang, Di},
+  journal = {arXiv preprint arXiv:2407.03168},
+  year    = {2024}
+}
+```
--- a/requirements.txt
+++ b/requirements.txt
+-r requirements_base.txt
+onnxruntime-gpu==1.18.0
--- a/requirements_base.txt
+++ b/requirements_base.txt
+# --extra-index-url https://download.pytorch.org/whl/cu118
+# torch==2.3.0
+# torchvision==0.18.0
+# torchaudio==2.3.0
+numpy==1.26.4
+pyyaml==6.0.1
+opencv-python==4.10.0.84
+scipy==1.13.1
+imageio==2.34.2
+lmdb==1.4.1
+tqdm==4.66.4
+rich==13.7.1
+ffmpeg-python==0.2.0
+onnx==1.16.1
+scikit-image==0.24.0
+albumentations==1.4.10
+matplotlib==3.9.0
+imageio-ffmpeg==0.5.1
+tyro==0.8.5
+gradio==4.37.1
+pykalman==0.9.7
--- a/requirements_docker.txt
+++ b/requirements_docker.txt
+pyyaml==6.0.1
+imageio==2.34.2
+lmdb==1.4.1
+tqdm==4.66.4
+rich==13.7.1
+ffmpeg-python==0.2.0
+scikit-image==0.24.0
+albumentations==1.4.10
+matplotlib==3.9.0
+imageio-ffmpeg==0.5.1
+tyro==0.8.5
+gradio==4.37.1
+onnx==1.16.1
+pykalman==0.9.7
\ No newline at end of file
--- a/requirements_macOS.txt
+++ b/requirements_macOS.txt
+-r requirements_base.txt
+onnxruntime-silicon==1.16.3
--- a/speed.py
+++ b/speed.py
+# coding: utf-8
+"""
+Benchmark the inference speed of each module in LivePortrait.
+TODO: heavy GPT style, need to refactor
+"""
+import torch
+torch._dynamo.config.suppress_errors = True  # Suppress errors and fall back to eager execution
+import yaml
+import time
+import numpy as np
+from src.utils.helper import load_model, concat_feat
+from src.config.inference_config import InferenceConfig
+def initialize_inputs(batch_size=1, device_id=0):
+    """
+    Generate random input tensors and move them to GPU
+    """
+    feature_3d = torch.randn(batch_size, 32, 16, 64, 64).to(device_id).half()
+    kp_source = torch.randn(batch_size, 21, 3).to(device_id).half()
+    kp_driving = torch.randn(batch_size, 21, 3).to(device_id).half()
+    source_image = torch.randn(batch_size, 3, 256, 256).to(device_id).half()
+    generator_input = torch.randn(batch_size, 256, 64, 64).to(device_id).half()
+    eye_close_ratio = torch.randn(batch_size, 3).to(device_id).half()
+    lip_close_ratio = torch.randn(batch_size, 2).to(device_id).half()
+    feat_stitching = concat_feat(kp_source, kp_driving).half()
+    feat_eye = concat_feat(kp_source, eye_close_ratio).half()
+    feat_lip = concat_feat(kp_source, lip_close_ratio).half()
+    inputs = {
+        'feature_3d': feature_3d,
+        'kp_source': kp_source,
+        'kp_driving': kp_driving,
+        'source_image': source_image,
+        'generator_input': generator_input,
+        'feat_stitching': feat_stitching,
+        'feat_eye': feat_eye,
+        'feat_lip': feat_lip
+    }
+    return inputs
+def load_and_compile_models(cfg, model_config):
+    """
+    Load and compile models for inference
+    """
+    appearance_feature_extractor = load_model(cfg.checkpoint_F, model_config, cfg.device_id, 'appearance_feature_extractor')
+    motion_extractor = load_model(cfg.checkpoint_M, model_config, cfg.device_id, 'motion_extractor')
+    warping_module = load_model(cfg.checkpoint_W, model_config, cfg.device_id, 'warping_module')
+    spade_generator = load_model(cfg.checkpoint_G, model_config, cfg.device_id, 'spade_generator')
+    stitching_retargeting_module = load_model(cfg.checkpoint_S, model_config, cfg.device_id, 'stitching_retargeting_module')
+    models_with_params = [
+        ('Appearance Feature Extractor', appearance_feature_extractor),
+        ('Motion Extractor', motion_extractor),
+        ('Warping Network', warping_module),
+        ('SPADE Decoder', spade_generator)
+    ]
+    compiled_models = {}
+    for name, model in models_with_params:
+        model = model.half()
+        model = torch.compile(model, mode='max-autotune')  # Optimize for inference
+        model.eval()  # Switch to evaluation mode
+        compiled_models[name] = model
+    retargeting_models = ['stitching', 'eye', 'lip']
+    for retarget in retargeting_models:
+        module = stitching_retargeting_module[retarget].half()
+        module = torch.compile(module, mode='max-autotune')  # Optimize for inference
+        module.eval()  # Switch to evaluation mode
+        stitching_retargeting_module[retarget] = module
+    return compiled_models, stitching_retargeting_module
+def warm_up_models(compiled_models, stitching_retargeting_module, inputs):
+    """
+    Warm up models to prepare them for benchmarking
+    """
+    print("Warm up start!")
+    with torch.no_grad():
+        for _ in range(10):
+            compiled_models['Appearance Feature Extractor'](inputs['source_image'])
+            compiled_models['Motion Extractor'](inputs['source_image'])
+            compiled_models['Warping Network'](inputs['feature_3d'], inputs['kp_driving'], inputs['kp_source'])
+            compiled_models['SPADE Decoder'](inputs['generator_input'])  # Adjust input as required
+            stitching_retargeting_module['stitching'](inputs['feat_stitching'])
+            stitching_retargeting_module['eye'](inputs['feat_eye'])
+            stitching_retargeting_module['lip'](inputs['feat_lip'])
+    print("Warm up end!")
+def measure_inference_times(compiled_models, stitching_retargeting_module, inputs):
+    """
+    Measure inference times for each model
+    """
+    times = {name: [] for name in compiled_models.keys()}
+    times['Stitching and Retargeting Modules'] = []
+    overall_times = []
+    with torch.no_grad():
+        for _ in range(100):
+            torch.cuda.synchronize()
+            overall_start = time.time()
+            start = time.time()
+            compiled_models['Appearance Feature Extractor'](inputs['source_image'])
+            torch.cuda.synchronize()
+            times['Appearance Feature Extractor'].append(time.time() - start)
+            start = time.time()
+            compiled_models['Motion Extractor'](inputs['source_image'])
+            torch.cuda.synchronize()
+            times['Motion Extractor'].append(time.time() - start)
+            start = time.time()
+            compiled_models['Warping Network'](inputs['feature_3d'], inputs['kp_driving'], inputs['kp_source'])
+            torch.cuda.synchronize()
+            times['Warping Network'].append(time.time() - start)
+            start = time.time()
+            compiled_models['SPADE Decoder'](inputs['generator_input'])  # Adjust input as required
+            torch.cuda.synchronize()
+            times['SPADE Decoder'].append(time.time() - start)
+            start = time.time()
+            stitching_retargeting_module['stitching'](inputs['feat_stitching'])
+            stitching_retargeting_module['eye'](inputs['feat_eye'])
+            stitching_retargeting_module['lip'](inputs['feat_lip'])
+            torch.cuda.synchronize()
+            times['Stitching and Retargeting Modules'].append(time.time() - start)
+            overall_times.append(time.time() - overall_start)
+    return times, overall_times
+def print_benchmark_results(compiled_models, stitching_retargeting_module, retargeting_models, times, overall_times):
+    """
+    Print benchmark results with average and standard deviation of inference times
+    """
+    average_times = {name: np.mean(times[name]) * 1000 for name in times.keys()}
+    std_times = {name: np.std(times[name]) * 1000 for name in times.keys()}
+    for name, model in compiled_models.items():
+        num_params = sum(p.numel() for p in model.parameters())
+        num_params_in_millions = num_params / 1e6
+        print(f"Number of parameters for {name}: {num_params_in_millions:.2f} M")
+    for index, retarget in enumerate(retargeting_models):
+        num_params = sum(p.numel() for p in stitching_retargeting_module[retarget].parameters())
+        num_params_in_millions = num_params / 1e6
+        print(f"Number of parameters for part_{index} in Stitching and Retargeting Modules: {num_params_in_millions:.2f} M")
+    for name, avg_time in average_times.items():
+        std_time = std_times[name]
+        print(f"Average inference time for {name} over 100 runs: {avg_time:.2f} ms (std: {std_time:.2f} ms)")
+def main():
+    """
+    Main function to benchmark speed and model parameters
+    """
+    # Load configuration
+    cfg = InferenceConfig()
+    model_config_path = cfg.models_config
+    with open(model_config_path, 'r') as file:
+        model_config = yaml.safe_load(file)
+    # Sample input tensors
+    inputs = initialize_inputs(device_id = cfg.device_id)
+    # Load and compile models
+    compiled_models, stitching_retargeting_module = load_and_compile_models(cfg, model_config)
+    # Warm up models
+    warm_up_models(compiled_models, stitching_retargeting_module, inputs)
+    # Measure inference times
+    times, overall_times = measure_inference_times(compiled_models, stitching_retargeting_module, inputs)
+    # Print benchmark results
+    print_benchmark_results(compiled_models, stitching_retargeting_module, ['stitching', 'eye', 'lip'], times, overall_times)
+if __name__ == "__main__":
+    main()
--- a/src/config/__init__.py
+++ b/src/config/__init__.py
--- a/src/config/argument_config.py
+++ b/src/config/argument_config.py
+# coding: utf-8
+"""
+All configs for user
+"""
+from dataclasses import dataclass
+import tyro
+from typing_extensions import Annotated
+from typing import Optional
+from .base_config import PrintableConfig, make_abs_path
+@dataclass(repr=False)  # use repr from PrintableConfig
+class ArgumentConfig(PrintableConfig):
+    ########## input arguments ##########
+    source: Annotated[str, tyro.conf.arg(aliases=["-s"])] = make_abs_path('../../assets/examples/source/s0.jpg')  # path to the source portrait or video
+    driving:  Annotated[str, tyro.conf.arg(aliases=["-d"])] = make_abs_path('../../assets/examples/driving/d0.mp4')  # path to driving video or template (.pkl format)
+    output_dir: Annotated[str, tyro.conf.arg(aliases=["-o"])] = 'animations/'  # directory to save output video
+    ########## inference arguments ##########
+    flag_use_half_precision: bool = True  # whether to use half precision (FP16). If black boxes appear, it might be due to GPU incompatibility; set to False.
+    flag_crop_driving_video: bool = False  # whether to crop the driving video, if the given driving info is a video
+    device_id: int = 0  # gpu device id
+    flag_force_cpu: bool = False  # force cpu inference, WIP!
+    flag_normalize_lip: bool = True  # whether to let the lip to close state before animation, only take effect when flag_eye_retargeting and flag_lip_retargeting is False
+    flag_source_video_eye_retargeting: bool = False  # when the input is a source video, whether to let the eye-open scalar of each frame to be the same as the first source frame before the animation, only take effect when flag_eye_retargeting and flag_lip_retargeting is False, may cause the inter-frame jittering
+    flag_video_editing_head_rotation: bool = False  # when the input is a source video, whether to inherit the relative head rotation from the driving video
+    flag_eye_retargeting: bool = False  # not recommend to be True, WIP
+    flag_lip_retargeting: bool = False  # not recommend to be True, WIP
+    flag_stitching: bool = True  # recommend to True if head movement is small, False if head movement is large
+    flag_relative_motion: bool = True  # whether to use relative motion
+    flag_pasteback: bool = True  # whether to paste-back/stitch the animated face cropping from the face-cropping space to the original image space
+    flag_do_crop: bool = True  # whether to crop the source portrait or video to the face-cropping space
+    driving_smooth_observation_variance: float = 1e-7  # smooth strength scalar for the animated video when the input is a source video, the larger the number, the smoother the animated video; too much smoothness would result in loss of motion accuracy
+    ########## source crop arguments ##########
+    scale: float = 2.3  # the ratio of face area is smaller if scale is larger
+    vx_ratio: float = 0  # the ratio to move the face to left or right in cropping space
+    vy_ratio: float = -0.125  # the ratio to move the face to up or down in cropping space
+    flag_do_rot: bool = True  # whether to conduct the rotation when flag_do_crop is True
+    ########## driving crop arguments ##########
+    scale_crop_driving_video: float = 2.2  # scale factor for cropping driving video
+    vx_ratio_crop_driving_video: float = 0.  # adjust y offset
+    vy_ratio_crop_driving_video: float = -0.1  # adjust x offset
+    ########## gradio arguments ##########
+    server_port: Annotated[int, tyro.conf.arg(aliases=["-p"])] = 8890  # port for gradio server
+    share: bool = False  # whether to share the server to public
+    server_name: Optional[str] = "127.0.0.1"  # set the local server name, "0.0.0.0" to broadcast all
+    flag_do_torch_compile: bool = False  # whether to use torch.compile to accelerate generation
\ No newline at end of file
--- a/src/config/base_config.py
+++ b/src/config/base_config.py
+# coding: utf-8
+"""
+pretty printing class
+"""
+from __future__ import annotations
+import os.path as osp
+from typing import Tuple
+def make_abs_path(fn):
+    return osp.join(osp.dirname(osp.realpath(__file__)), fn)
+class PrintableConfig:  # pylint: disable=too-few-public-methods
+    """Printable Config defining str function"""
+    def __repr__(self):
+        lines = [self.__class__.__name__ + ":"]
+        for key, val in vars(self).items():
+            if isinstance(val, Tuple):
+                flattened_val = "["
+                for item in val:
+                    flattened_val += str(item) + "\n"
+                flattened_val = flattened_val.rstrip("\n")
+                val = flattened_val + "]"
+            lines += f"{key}: {str(val)}".split("\n")
+        return "\n    ".join(lines)