Initial commit

25d2752f · yongshk · 25d2752f · 25d2752f · 25d2752f · 25d2752f
Commit 25d2752f authored May 29, 2025 by yongshk
20 changed files
--- a/candle-book/src/inference/inference.md
+++ b/candle-book/src/inference/inference.md
+# Running a model
+In order to run an existing model, you will need to download and use existing weights.
+Most models are already available on https://huggingface.co/ in [`safetensors`](https://github.com/huggingface/safetensors) format.
+Let's get started by running an old model : `bert-base-uncased`.
--- a/candle-book/src/lib.rs
+++ b/candle-book/src/lib.rs
+#[cfg(test)]
+pub mod simplified;
+#[cfg(test)]
+mod tests {
+    use anyhow::Result;
+    use candle::{DType, Device, Tensor};
+    use parquet::file::reader::SerializedFileReader;
+    // NOTE: Waiting on https://github.com/rust-lang/mdBook/pull/1856
+    #[rustfmt::skip]
+    #[tokio::test]
+    async fn book_hub_1() {
+// ANCHOR: book_hub_1
+use candle::Device;
+use hf_hub::api::tokio::Api;
+let api = Api::new().unwrap();
+let repo = api.model("bert-base-uncased".to_string());
+let weights_filename = repo.get("model.safetensors").await.unwrap();
+let weights = candle::safetensors::load(weights_filename, &Device::Cpu).unwrap();
+// ANCHOR_END: book_hub_1
+        assert_eq!(weights.len(), 206);
+    }
+    #[rustfmt::skip]
+    #[test]
+    fn book_hub_2() {
+        {
+// ANCHOR: book_hub_2
+use candle::Device;
+use hf_hub::api::sync::Api;
+use memmap2::Mmap;
+use std::fs;
+let api = Api::new().unwrap();
+let repo = api.model("bert-base-uncased".to_string());
+let weights_filename = repo.get("model.safetensors").unwrap();
+let file = fs::File::open(weights_filename).unwrap();
+let mmap = unsafe { Mmap::map(&file).unwrap() };
+let weights = candle::safetensors::load_buffer(&mmap[..], &Device::Cpu).unwrap();
+// ANCHOR_END: book_hub_2
+        assert_eq!(weights.len(), 206);
+    }
+    // #[rustfmt::skip]
+    // #[test]
+    // fn book_hub_3() {
+    {
+// ANCHOR: book_hub_3
+use candle::{DType, Device, Tensor};
+use hf_hub::api::sync::Api;
+use memmap2::Mmap;
+use safetensors::slice::IndexOp;
+use safetensors::SafeTensors;
+use std::fs;
+let api = Api::new().unwrap();
+let repo = api.model("bert-base-uncased".to_string());
+let weights_filename = repo.get("model.safetensors").unwrap();
+let file = fs::File::open(weights_filename).unwrap();
+let mmap = unsafe { Mmap::map(&file).unwrap() };
+// Use safetensors directly
+let tensors = SafeTensors::deserialize(&mmap[..]).unwrap();
+let view = tensors
+    .tensor("bert.encoder.layer.0.attention.self.query.weight")
+    .unwrap();
+// We're going to load shard with rank 1, within a world_size of 4
+// We're going to split along dimension 0 doing VIEW[start..stop, :]
+let rank = 1;
+let world_size = 4;
+let dim = 0;
+let dtype = view.dtype();
+let mut tp_shape = view.shape().to_vec();
+let size = tp_shape[0];
+if size % world_size != 0 {
+    panic!("The dimension is not divisble by `world_size`");
+}
+let block_size = size / world_size;
+let start = rank * block_size;
+let stop = (rank + 1) * block_size;
+// Everything is expressed in tensor dimension
+// bytes offsets is handled automatically for safetensors.
+let iterator = view.slice(start..stop).unwrap();
+tp_shape[dim] = block_size;
+// Convert safetensors Dtype to candle DType
+let dtype: DType = dtype.try_into().unwrap();
+// TODO: Implement from_buffer_iterator so we can skip the extra CPU alloc.
+let raw: Vec<u8> = iterator.into_iter().flatten().cloned().collect();
+let tp_tensor = Tensor::from_raw_buffer(&raw, dtype, &tp_shape, &Device::Cpu).unwrap();
+// ANCHOR_END: book_hub_3
+        assert_eq!(view.shape(), &[768, 768]);
+        assert_eq!(tp_tensor.dims(), &[192, 768]);
+    }
+}
+    #[rustfmt::skip]
+    #[test]
+    fn book_training_1() -> Result<()>{
+// ANCHOR: book_training_1
+use hf_hub::{api::sync::Api, Repo, RepoType};
+let dataset_id = "mnist".to_string();
+let api = Api::new()?;
+let repo = Repo::with_revision(
+    dataset_id,
+    RepoType::Dataset,
+    "refs/convert/parquet".to_string(),
+);
+let repo = api.repo(repo);
+let test_parquet_filename = repo.get("mnist/test/0000.parquet")?;
+let train_parquet_filename = repo.get("mnist/train/0000.parquet")?;
+let test_parquet = SerializedFileReader::new(std::fs::File::open(test_parquet_filename)?)?;
+let train_parquet = SerializedFileReader::new(std::fs::File::open(train_parquet_filename)?)?;
+// ANCHOR_END: book_training_1
+// Ignore unused
+let _train = train_parquet;
+// ANCHOR: book_training_2
+for row in test_parquet {
+    for (idx, (name, field)) in row?.get_column_iter().enumerate() {
+        println!("Column id {idx}, name {name}, value {field}");
+    }
+}
+// ANCHOR_END: book_training_2
+let test_parquet_filename = repo.get("mnist/test/0000.parquet")?;
+let train_parquet_filename = repo.get("mnist/train/0000.parquet")?;
+let test_parquet = SerializedFileReader::new(std::fs::File::open(test_parquet_filename)?)?;
+let train_parquet = SerializedFileReader::new(std::fs::File::open(train_parquet_filename)?)?;
+// ANCHOR: book_training_3
+let test_samples = 10_000;
+let mut test_buffer_images: Vec<u8> = Vec::with_capacity(test_samples * 784);
+let mut test_buffer_labels: Vec<u8> = Vec::with_capacity(test_samples);
+for row in test_parquet{
+    for (_name, field) in row?.get_column_iter() {
+        if let parquet::record::Field::Group(subrow) = field {
+            for (_name, field) in subrow.get_column_iter() {
+                if let parquet::record::Field::Bytes(value) = field {
+                    let image = image::load_from_memory(value.data()).unwrap();
+                    test_buffer_images.extend(image.to_luma8().as_raw());
+                }
+            }
+        }else if let parquet::record::Field::Long(label) = field {
+            test_buffer_labels.push(*label as u8);
+        }
+    }
+}
+let test_images = (Tensor::from_vec(test_buffer_images, (test_samples, 784), &Device::Cpu)?.to_dtype(DType::F32)? / 255.)?;
+let test_labels = Tensor::from_vec(test_buffer_labels, (test_samples, ), &Device::Cpu)?;
+let train_samples = 60_000;
+let mut train_buffer_images: Vec<u8> = Vec::with_capacity(train_samples * 784);
+let mut train_buffer_labels: Vec<u8> = Vec::with_capacity(train_samples);
+for row in train_parquet{
+    for (_name, field) in row?.get_column_iter() {
+        if let parquet::record::Field::Group(subrow) = field {
+            for (_name, field) in subrow.get_column_iter() {
+                if let parquet::record::Field::Bytes(value) = field {
+                    let image = image::load_from_memory(value.data()).unwrap();
+                    train_buffer_images.extend(image.to_luma8().as_raw());
+                }
+            }
+        }else if let parquet::record::Field::Long(label) = field {
+            train_buffer_labels.push(*label as u8);
+        }
+    }
+}
+let train_images = (Tensor::from_vec(train_buffer_images, (train_samples, 784), &Device::Cpu)?.to_dtype(DType::F32)? / 255.)?;
+let train_labels = Tensor::from_vec(train_buffer_labels, (train_samples, ), &Device::Cpu)?;
+let mnist = candle_datasets::vision::Dataset {
+    train_images,
+    train_labels,
+    test_images,
+    test_labels,
+    labels: 10,
+};
+// ANCHOR_END: book_training_3
+assert_eq!(mnist.test_images.dims(), &[10_000, 784]);
+assert_eq!(mnist.test_labels.dims(), &[10_000]);
+assert_eq!(mnist.train_images.dims(), &[60_000, 784]);
+assert_eq!(mnist.train_labels.dims(), &[60_000]);
+Ok(())
+    }
+}
--- a/candle-book/src/simplified.rs
+++ b/candle-book/src/simplified.rs
+//! #A simplified example in Rust of training a neural network and then using it based on the Candle Framework by Hugging Face.
+//! Author: Evgeny Igumnov 2023 igumnovnsk@gmail.com
+//! This program implements a neural network to predict the winner of the second round of elections based on the results of the first round.
+//!
+//! ##Basic moments:
+//!
+//! A multilayer perceptron with two hidden layers is used. The first hidden layer has 4 neurons, the second has 2 neurons.
+//! The input is a vector of 2 numbers - the percentage of votes for the first and second candidates in the first stage.
+//! The output is the number 0 or 1, where 1 means that the first candidate will win in the second stage, 0 means that he will lose.
+//! For training, samples with real data on the results of the first and second stages of different elections are used.
+//! The model is trained by backpropagation using gradient descent and the cross-entropy loss function.
+//! Model parameters (weights of neurons) are initialized randomly, then optimized during training.
+//! After training, the model is tested on a deferred sample to evaluate the accuracy.
+//! If the accuracy on the test set is below 100%, the model is considered underfit and the learning process is repeated.
+//! Thus, this neural network learns to find hidden relationships between the results of the first and second rounds of voting in order to make predictions for new data.
+#[rustfmt::skip]
+mod tests {
+use candle::{DType, Result, Tensor, D, Device};
+use candle_nn::{loss, ops, Linear, Module, VarBuilder, VarMap, Optimizer};
+// ANCHOR: book_training_simplified1
+const VOTE_DIM: usize = 2;
+const RESULTS: usize = 1;
+const EPOCHS: usize = 10;
+const LAYER1_OUT_SIZE: usize = 4;
+const LAYER2_OUT_SIZE: usize = 2;
+const LEARNING_RATE: f64 = 0.05;
+#[derive(Clone)]
+pub struct Dataset {
+    pub train_votes: Tensor,
+    pub train_results: Tensor,
+    pub test_votes: Tensor,
+    pub test_results: Tensor,
+}
+struct MultiLevelPerceptron {
+    ln1: Linear,
+    ln2: Linear,
+    ln3: Linear,
+}
+impl MultiLevelPerceptron {
+    fn new(vs: VarBuilder) -> Result<Self> {
+        let ln1 = candle_nn::linear(VOTE_DIM, LAYER1_OUT_SIZE, vs.pp("ln1"))?;
+        let ln2 = candle_nn::linear(LAYER1_OUT_SIZE, LAYER2_OUT_SIZE, vs.pp("ln2"))?;
+        let ln3 = candle_nn::linear(LAYER2_OUT_SIZE, RESULTS + 1, vs.pp("ln3"))?;
+        Ok(Self { ln1, ln2, ln3 })
+    }
+    fn forward(&self, xs: &Tensor) -> Result<Tensor> {
+        let xs = self.ln1.forward(xs)?;
+        let xs = xs.relu()?;
+        let xs = self.ln2.forward(&xs)?;
+        let xs = xs.relu()?;
+        self.ln3.forward(&xs)
+    }
+}
+// ANCHOR_END: book_training_simplified1
+// ANCHOR: book_training_simplified3
+#[tokio::test]
+async fn simplified() -> anyhow::Result<()> {
+    let dev = Device::cuda_if_available(0)?;
+    let train_votes_vec: Vec<u32> = vec![
+        15, 10,
+        10, 15,
+        5, 12,
+        30, 20,
+        16, 12,
+        13, 25,
+        6, 14,
+        31, 21,
+    ];
+    let train_votes_tensor = Tensor::from_vec(train_votes_vec.clone(), (train_votes_vec.len() / VOTE_DIM, VOTE_DIM), &dev)?.to_dtype(DType::F32)?;
+    let train_results_vec: Vec<u32> = vec![
+        1,
+        0,
+        0,
+        1,
+        1,
+        0,
+        0,
+        1,
+    ];
+    let train_results_tensor = Tensor::from_vec(train_results_vec, train_votes_vec.len() / VOTE_DIM, &dev)?;
+    let test_votes_vec: Vec<u32> = vec![
+        13, 9,
+        8, 14,
+        3, 10,
+    ];
+    let test_votes_tensor = Tensor::from_vec(test_votes_vec.clone(), (test_votes_vec.len() / VOTE_DIM, VOTE_DIM), &dev)?.to_dtype(DType::F32)?;
+    let test_results_vec: Vec<u32> = vec![
+        1,
+        0,
+        0,
+    ];
+    let test_results_tensor = Tensor::from_vec(test_results_vec.clone(), test_results_vec.len(), &dev)?;
+    let m = Dataset {
+        train_votes: train_votes_tensor,
+        train_results: train_results_tensor,
+        test_votes: test_votes_tensor,
+        test_results: test_results_tensor,
+    };
+    let trained_model: MultiLevelPerceptron;
+    loop {
+        println!("Trying to train neural network.");
+        match train(m.clone(), &dev) {
+            Ok(model) => {
+                trained_model = model;
+                break;
+            },
+            Err(e) => {
+                println!("Error: {}", e);
+                continue;
+            }
+        }
+    }
+    let real_world_votes: Vec<u32> = vec![
+        13, 22,
+    ];
+    let tensor_test_votes = Tensor::from_vec(real_world_votes.clone(), (1, VOTE_DIM), &dev)?.to_dtype(DType::F32)?;
+    let final_result = trained_model.forward(&tensor_test_votes)?;
+    let result = final_result
+        .argmax(D::Minus1)?
+        .to_dtype(DType::F32)?
+        .get(0).map(|x| x.to_scalar::<f32>())??;
+    println!("real_life_votes: {:?}", real_world_votes);
+    println!("neural_network_prediction_result: {:?}", result);
+    Ok(())
+}
+// ANCHOR_END: book_training_simplified3
+// ANCHOR: book_training_simplified2
+fn train(m: Dataset, dev: &Device) -> anyhow::Result<MultiLevelPerceptron> {
+    let train_results = m.train_results.to_device(dev)?;
+    let train_votes = m.train_votes.to_device(dev)?;
+    let varmap = VarMap::new();
+    let vs = VarBuilder::from_varmap(&varmap, DType::F32, dev);
+    let model = MultiLevelPerceptron::new(vs.clone())?;
+    let mut sgd = candle_nn::SGD::new(varmap.all_vars(), LEARNING_RATE)?;
+    let test_votes = m.test_votes.to_device(dev)?;
+    let test_results = m.test_results.to_device(dev)?;
+    let mut final_accuracy: f32 = 0.0;
+    for epoch in 1..EPOCHS + 1 {
+        let logits = model.forward(&train_votes)?;
+        let log_sm = ops::log_softmax(&logits, D::Minus1)?;
+        let loss = loss::nll(&log_sm, &train_results)?;
+        sgd.backward_step(&loss)?;
+        let test_logits = model.forward(&test_votes)?;
+        let sum_ok = test_logits
+            .argmax(D::Minus1)?
+            .eq(&test_results)?
+            .to_dtype(DType::F32)?
+            .sum_all()?
+            .to_scalar::<f32>()?;
+        let test_accuracy = sum_ok / test_results.dims1()? as f32;
+        final_accuracy = 100. * test_accuracy;
+        println!("Epoch: {epoch:3} Train loss: {:8.5} Test accuracy: {:5.2}%",
+                 loss.to_scalar::<f32>()?,
+                 final_accuracy
+        );
+        if final_accuracy == 100.0 {
+            break;
+        }
+    }
+    if final_accuracy < 100.0 {
+        Err(anyhow::Error::msg("The model is not trained well enough."))
+    } else {
+        Ok(model)
+    }
+}
+// ANCHOR_END: book_training_simplified2
+}
--- a/candle-book/src/training/finetuning.md
+++ b/candle-book/src/training/finetuning.md
+# Fine-tuning
--- a/candle-book/src/training/mnist.md
+++ b/candle-book/src/training/mnist.md
+# MNIST
+So we now have downloaded the MNIST parquet files, let's put them in a simple struct.
+```rust,ignore
+{{#include ../lib.rs:book_training_3}}
+```
+The parsing of the file and putting it into single tensors requires the dataset to fit the entire memory.
+It is quite rudimentary, but simple enough for a small dataset like MNIST.
--- a/candle-book/src/training/serialization.md
+++ b/candle-book/src/training/serialization.md
+# Serialization
--- a/candle-book/src/training/simplified.md
+++ b/candle-book/src/training/simplified.md
+# Simplified
+## How its works
+This program implements a neural network to predict the winner of the second round of elections based on the results of the first round.
+Basic moments:
+1. A multilayer perceptron with two hidden layers is used. The first hidden layer has 4 neurons, the second has 2 neurons.
+2. The input is a vector of 2 numbers - the percentage of votes for the first and second candidates in the first stage.
+3. The output is the number 0 or 1, where 1 means that the first candidate will win in the second stage, 0 means that he will lose.
+4. For training, samples with real data on the results of the first and second stages of different elections are used.
+5. The model is trained by backpropagation using gradient descent and the cross-entropy loss function.
+6. Model parameters (weights of neurons) are initialized randomly, then optimized during training.
+7. After training, the model is tested on a deferred sample to evaluate the accuracy.
+8. If the accuracy on the test set is below 100%, the model is considered underfit and the learning process is repeated.
+Thus, this neural network learns to find hidden relationships between the results of the first and second rounds of voting in order to make predictions for new data.
+```rust,ignore
+{{#include ../simplified.rs:book_training_simplified1}}
+```
+```rust,ignore
+{{#include ../simplified.rs:book_training_simplified2}}
+```
+```rust,ignore
+{{#include ../simplified.rs:book_training_simplified3}}
+```
+## Example output
+```bash
+Trying to train neural network.
+Epoch:   1 Train loss:  4.42555 Test accuracy:  0.00%
+Epoch:   2 Train loss:  0.84677 Test accuracy: 33.33%
+Epoch:   3 Train loss:  2.54335 Test accuracy: 33.33%
+Epoch:   4 Train loss:  0.37806 Test accuracy: 33.33%
+Epoch:   5 Train loss:  0.36647 Test accuracy: 100.00%
+real_life_votes: [13, 22]
+neural_network_prediction_result: 0.0
+```
--- a/candle-book/src/training/training.md
+++ b/candle-book/src/training/training.md
+# Training
+Training starts with data. We're going to use the huggingface hub and 
+start with the Hello world dataset of machine learning, MNIST.
+Let's start with downloading `MNIST` from [huggingface](https://huggingface.co/datasets/mnist).
+This requires [`hf-hub`](https://github.com/huggingface/hf-hub).
+```bash
+cargo add hf-hub
+```
+This is going to be very hands-on for now.
+```rust,ignore
+{{#include ../../../candle-examples/src/lib.rs:book_training_1}}
+```
+This uses the standardized `parquet` files from the `refs/convert/parquet` branch on every dataset.
+Our handles are now [`parquet::file::serialized_reader::SerializedFileReader`].
+We can inspect the content of the files with:
+```rust,ignore
+{{#include ../../../candle-examples/src/lib.rs:book_training_2}}
+```
+You should see something like:
+```bash
+Column id 1, name label, value 6
+Column id 0, name image, value {bytes: [137, ....]
+Column id 1, name label, value 8
+Column id 0, name image, value {bytes: [137, ....]
+```
+So each row contains 2 columns (image, label) with image being saved as bytes.
+Let's put them into a useful struct.
--- a/candle-core/Cargo.toml
+++ b/candle-core/Cargo.toml
+[package]
+name = "candle-core"
+version.workspace = true
+edition.workspace = true
+description.workspace = true
+repository.workspace = true
+keywords.workspace = true
+categories.workspace = true
+license.workspace = true
+readme = "README.md"
+[dependencies]
+accelerate-src = { workspace = true, optional = true }
+byteorder = { workspace = true }
+candle-kernels = { workspace = true, optional = true }
+candle-metal-kernels = { workspace = true, optional = true }
+metal = { workspace = true, optional = true}
+cudarc = { workspace = true, optional = true }
+gemm = { workspace = true }
+half = { workspace = true }
+intel-mkl-src = { workspace = true, optional = true }
+libc = { workspace = true, optional = true }
+memmap2 = { workspace = true }
+num-traits = { workspace = true }
+num_cpus = { workspace = true }
+rand = { workspace = true }
+rand_distr = { workspace = true }
+rayon = { workspace = true }
+safetensors = { workspace = true }
+thiserror = { workspace = true }
+yoke = { workspace = true }
+zip = { workspace = true }
+[dev-dependencies]
+anyhow = { workspace = true }
+clap = { workspace = true }
+criterion = { workspace = true }
+[features]
+default = []
+cuda = ["cudarc", "dep:candle-kernels"]
+cudnn = ["cuda", "cudarc/cudnn"]
+mkl = ["dep:libc", "dep:intel-mkl-src", "intel-mkl-src/mkl-static-lp64-iomp"]
+mkl-dynamic = ["dep:libc", "dep:intel-mkl-src", "intel-mkl-src/mkl-dynamic-lp64-iomp"]
+accelerate = ["dep:libc", "dep:accelerate-src"]
+metal = ["dep:metal", "dep:candle-metal-kernels"]
+[[bench]]
+name = "bench_main"
+harness = false
--- a/candle-core/LICENSE
+++ b/candle-core/LICENSE
+                                 Apache License
+                           Version 2.0, January 2004
+                        http://www.apache.org/licenses/
+   TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+   1. Definitions.
+      "License" shall mean the terms and conditions for use, reproduction,
+      and distribution as defined by Sections 1 through 9 of this document.
+      "Licensor" shall mean the copyright owner or entity authorized by
+      the copyright owner that is granting the License.
+      "Legal Entity" shall mean the union of the acting entity and all
+      other entities that control, are controlled by, or are under common
+      control with that entity. For the purposes of this definition,
+      "control" means (i) the power, direct or indirect, to cause the
+      direction or management of such entity, whether by contract or
+      otherwise, or (ii) ownership of fifty percent (50%) or more of the
+      outstanding shares, or (iii) beneficial ownership of such entity.
+      "You" (or "Your") shall mean an individual or Legal Entity
+      exercising permissions granted by this License.
+      "Source" form shall mean the preferred form for making modifications,
+      including but not limited to software source code, documentation
+      source, and configuration files.
+      "Object" form shall mean any form resulting from mechanical
+      transformation or translation of a Source form, including but
+      not limited to compiled object code, generated documentation,
+      and conversions to other media types.
+      "Work" shall mean the work of authorship, whether in Source or
+      Object form, made available under the License, as indicated by a
+      copyright notice that is included in or attached to the work
+      (an example is provided in the Appendix below).
+      "Derivative Works" shall mean any work, whether in Source or Object
+      form, that is based on (or derived from) the Work and for which the
+      editorial revisions, annotations, elaborations, or other modifications
+      represent, as a whole, an original work of authorship. For the purposes
+      of this License, Derivative Works shall not include works that remain
+      separable from, or merely link (or bind by name) to the interfaces of,
+      the Work and Derivative Works thereof.
+      "Contribution" shall mean any work of authorship, including
+      the original version of the Work and any modifications or additions
+      to that Work or Derivative Works thereof, that is intentionally
+      submitted to Licensor for inclusion in the Work by the copyright owner
+      or by an individual or Legal Entity authorized to submit on behalf of
+      the copyright owner. For the purposes of this definition, "submitted"
+      means any form of electronic, verbal, or written communication sent
+      to the Licensor or its representatives, including but not limited to
+      communication on electronic mailing lists, source code control systems,
+      and issue tracking systems that are managed by, or on behalf of, the
+      Licensor for the purpose of discussing and improving the Work, but
+      excluding communication that is conspicuously marked or otherwise
+      designated in writing by the copyright owner as "Not a Contribution."
+      "Contributor" shall mean Licensor and any individual or Legal Entity
+      on behalf of whom a Contribution has been received by Licensor and
+      subsequently incorporated within the Work.
+   2. Grant of Copyright License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      copyright license to reproduce, prepare Derivative Works of,
+      publicly display, publicly perform, sublicense, and distribute the
+      Work and such Derivative Works in Source or Object form.
+   3. Grant of Patent License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      (except as stated in this section) patent license to make, have made,
+      use, offer to sell, sell, import, and otherwise transfer the Work,
+      where such license applies only to those patent claims licensable
+      by such Contributor that are necessarily infringed by their
+      Contribution(s) alone or by combination of their Contribution(s)
+      with the Work to which such Contribution(s) was submitted. If You
+      institute patent litigation against any entity (including a
+      cross-claim or counterclaim in a lawsuit) alleging that the Work
+      or a Contribution incorporated within the Work constitutes direct
+      or contributory patent infringement, then any patent licenses
+      granted to You under this License for that Work shall terminate
+      as of the date such litigation is filed.
+   4. Redistribution. You may reproduce and distribute copies of the
+      Work or Derivative Works thereof in any medium, with or without
+      modifications, and in Source or Object form, provided that You
+      meet the following conditions:
+      (a) You must give any other recipients of the Work or
+          Derivative Works a copy of this License; and
+      (b) You must cause any modified files to carry prominent notices
+          stating that You changed the files; and
+      (c) You must retain, in the Source form of any Derivative Works
+          that You distribute, all copyright, patent, trademark, and
+          attribution notices from the Source form of the Work,
+          excluding those notices that do not pertain to any part of
+          the Derivative Works; and
+      (d) If the Work includes a "NOTICE" text file as part of its
+          distribution, then any Derivative Works that You distribute must
+          include a readable copy of the attribution notices contained
+          within such NOTICE file, excluding those notices that do not
+          pertain to any part of the Derivative Works, in at least one
+          of the following places: within a NOTICE text file distributed
+          as part of the Derivative Works; within the Source form or
+          documentation, if provided along with the Derivative Works; or,
+          within a display generated by the Derivative Works, if and
+          wherever such third-party notices normally appear. The contents
+          of the NOTICE file are for informational purposes only and
+          do not modify the License. You may add Your own attribution
+          notices within Derivative Works that You distribute, alongside
+          or as an addendum to the NOTICE text from the Work, provided
+          that such additional attribution notices cannot be construed
+          as modifying the License.
+      You may add Your own copyright statement to Your modifications and
+      may provide additional or different license terms and conditions
+      for use, reproduction, or distribution of Your modifications, or
+      for any such Derivative Works as a whole, provided Your use,
+      reproduction, and distribution of the Work otherwise complies with
+      the conditions stated in this License.
+   5. Submission of Contributions. Unless You explicitly state otherwise,
+      any Contribution intentionally submitted for inclusion in the Work
+      by You to the Licensor shall be under the terms and conditions of
+      this License, without any additional terms or conditions.
+      Notwithstanding the above, nothing herein shall supersede or modify
+      the terms of any separate license agreement you may have executed
+      with Licensor regarding such Contributions.
+   6. Trademarks. This License does not grant permission to use the trade
+      names, trademarks, service marks, or product names of the Licensor,
+      except as required for reasonable and customary use in describing the
+      origin of the Work and reproducing the content of the NOTICE file.
+   7. Disclaimer of Warranty. Unless required by applicable law or
+      agreed to in writing, Licensor provides the Work (and each
+      Contributor provides its Contributions) on an "AS IS" BASIS,
+      WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+      implied, including, without limitation, any warranties or conditions
+      of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+      PARTICULAR PURPOSE. You are solely responsible for determining the
+      appropriateness of using or redistributing the Work and assume any
+      risks associated with Your exercise of permissions under this License.
+   8. Limitation of Liability. In no event and under no legal theory,
+      whether in tort (including negligence), contract, or otherwise,
+      unless required by applicable law (such as deliberate and grossly
+      negligent acts) or agreed to in writing, shall any Contributor be
+      liable to You for damages, including any direct, indirect, special,
+      incidental, or consequential damages of any character arising as a
+      result of this License or out of the use or inability to use the
+      Work (including but not limited to damages for loss of goodwill,
+      work stoppage, computer failure or malfunction, or any and all
+      other commercial damages or losses), even if such Contributor
+      has been advised of the possibility of such damages.
+   9. Accepting Warranty or Additional Liability. While redistributing
+      the Work or Derivative Works thereof, You may choose to offer,
+      and charge a fee for, acceptance of support, warranty, indemnity,
+      or other liability obligations and/or rights consistent with this
+      License. However, in accepting such obligations, You may act only
+      on Your own behalf and on Your sole responsibility, not on behalf
+      of any other Contributor, and only if You agree to indemnify,
+      defend, and hold each Contributor harmless for any liability
+      incurred by, or claims asserted against, such Contributor by reason
+      of your accepting any such warranty or additional liability.
+   END OF TERMS AND CONDITIONS
+   APPENDIX: How to apply the Apache License to your work.
+      To apply the Apache License to your work, attach the following
+      boilerplate notice, with the fields enclosed by brackets "[]"
+      replaced with your own identifying information. (Don't include
+      the brackets!)  The text should be enclosed in the appropriate
+      comment syntax for the file format. We also recommend that a
+      file or class name and description of purpose be included on the
+      same "printed page" as the copyright notice for easier
+      identification within third-party archives.
+   Copyright [yyyy] [name of copyright owner]
+   Licensed under the Apache License, Version 2.0 (the "License");
+   you may not use this file except in compliance with the License.
+   You may obtain a copy of the License at
+       http://www.apache.org/licenses/LICENSE-2.0
+   Unless required by applicable law or agreed to in writing, software
+   distributed under the License is distributed on an "AS IS" BASIS,
+   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+   See the License for the specific language governing permissions and
+   limitations under the License.
--- a/candle-core/README.md
+++ b/candle-core/README.md
+# candle
+Minimalist ML framework for Rust
--- a/candle-core/benches/bench_main.rs
+++ b/candle-core/benches/bench_main.rs
+mod benchmarks;
+use criterion::criterion_main;
+criterion_main!(
+    benchmarks::affine::benches,
+    benchmarks::matmul::benches,
+    benchmarks::random::benches,
+    benchmarks::where_cond::benches,
+    benchmarks::conv_transpose2d::benches,
+);
--- a/candle-core/benches/benchmarks/affine.rs
+++ b/candle-core/benches/benchmarks/affine.rs
+use crate::benchmarks::{BenchDevice, BenchDeviceHandler};
+use candle_core::{DType, Device, Tensor};
+use criterion::{black_box, criterion_group, Criterion, Throughput};
+use std::time::Instant;
+fn run(a: &Tensor) {
+    a.affine(12.34, 56.78).unwrap();
+}
+fn run_affine_benchmark(c: &mut Criterion, device: &Device, dtype: DType, name: &str) {
+    let b = 1;
+    let m = 1024;
+    let k = 1024;
+    let tensor = Tensor::zeros((b, m, k), dtype, &device).unwrap();
+    let flops = b * m * k * dtype.size_in_bytes();
+    let mut group = c.benchmark_group(device.bench_name(name));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |b| {
+        b.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                run(black_box(&tensor));
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+}
+fn criterion_benchmark(c: &mut Criterion) {
+    let handler = BenchDeviceHandler::new().unwrap();
+    for device in handler.devices {
+        run_affine_benchmark(c, &device, DType::F32, "affine_f32");
+        run_affine_benchmark(c, &device, DType::F16, "affine_f16");
+        run_affine_benchmark(c, &device, DType::BF16, "affine_bf16");
+    }
+}
+criterion_group!(benches, criterion_benchmark);
--- a/candle-core/benches/benchmarks/conv_transpose2d.rs
+++ b/candle-core/benches/benchmarks/conv_transpose2d.rs
+use crate::benchmarks::{BenchDevice, BenchDeviceHandler};
+use candle_core::{DType, Device, Tensor};
+use criterion::{black_box, criterion_group, Criterion, Throughput};
+use std::time::Instant;
+fn run(
+    x: &Tensor,
+    k: &Tensor,
+    padding: usize,
+    output_padding: usize,
+    stride: usize,
+    dilation: usize,
+) {
+    x.conv_transpose2d(k, padding, output_padding, stride, dilation)
+        .unwrap();
+}
+fn run_benchmark(c: &mut Criterion, device: &Device, dtype: DType, name: &str) {
+    let t = Tensor::arange(0.0f32, 10000.0, device)
+        .unwrap()
+        .reshape((1, 4, 50, 50))
+        .unwrap()
+        .to_dtype(dtype)
+        .unwrap();
+    let kernel = Tensor::arange(0.0f32, 100.0, device)
+        .unwrap()
+        .reshape((4, 1, 5, 5))
+        .unwrap()
+        .to_dtype(dtype)
+        .unwrap();
+    let flops = t.dims().iter().product::<usize>() * dtype.size_in_bytes();
+    let mut group = c.benchmark_group(device.bench_name(name));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |b| {
+        b.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                run(black_box(&t), black_box(&kernel), 1, 0, 1, 2);
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+}
+fn criterion_benchmark(c: &mut Criterion) {
+    let handler = BenchDeviceHandler::new().unwrap();
+    for device in handler.devices {
+        run_benchmark(c, &device, DType::F32, "conv_transpose2d_f32");
+        run_benchmark(c, &device, DType::F16, "conv_transpose2d_f16");
+        run_benchmark(c, &device, DType::BF16, "conv_transpose2d_bf16");
+    }
+}
+criterion_group!(benches, criterion_benchmark);
--- a/candle-core/benches/benchmarks/matmul.rs
+++ b/candle-core/benches/benchmarks/matmul.rs
+use crate::benchmarks::{BenchDevice, BenchDeviceHandler};
+use candle_core::{DType, Device, Tensor};
+use criterion::{black_box, criterion_group, Criterion, Throughput};
+use std::time::Instant;
+fn run(a: &Tensor, b: &Tensor) {
+    a.matmul(&b.t().unwrap()).unwrap();
+}
+fn run_bench(c: &mut Criterion, device: &Device) {
+    let b = 1;
+    let m = 1;
+    let n = 2048;
+    let k = 2048;
+    let dtype = DType::F32;
+    let lhs = Tensor::zeros((b, m, k), dtype, device).unwrap();
+    let rhs = Tensor::zeros((b, n, k), dtype, device).unwrap();
+    let flops = b * m * n * k;
+    let mut group = c.benchmark_group(device.bench_name("matmul"));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |b| {
+        b.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                run(black_box(&lhs), black_box(&rhs));
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+}
+fn criterion_benchmark(c: &mut Criterion) {
+    let handler = BenchDeviceHandler::new().unwrap();
+    for device in handler.devices {
+        run_bench(c, &device);
+    }
+}
+criterion_group!(benches, criterion_benchmark);
--- a/candle-core/benches/benchmarks/mod.rs
+++ b/candle-core/benches/benchmarks/mod.rs
+pub(crate) mod affine;
+pub(crate) mod conv_transpose2d;
+pub(crate) mod matmul;
+pub(crate) mod random;
+pub(crate) mod where_cond;
+use candle_core::{Device, Result};
+pub(crate) trait BenchDevice {
+    fn sync(&self) -> Result<()>;
+    fn bench_name<S: Into<String>>(&self, name: S) -> String;
+}
+impl BenchDevice for Device {
+    fn sync(&self) -> Result<()> {
+        match self {
+            Device::Cpu => Ok(()),
+            Device::Cuda(device) => {
+                #[cfg(feature = "cuda")]
+                return Ok(device.synchronize()?);
+                #[cfg(not(feature = "cuda"))]
+                panic!("Cuda device without cuda feature enabled: {:?}", device)
+            }
+            Device::Metal(device) => {
+                #[cfg(feature = "metal")]
+                return Ok(device.wait_until_completed()?);
+                #[cfg(not(feature = "metal"))]
+                panic!("Metal device without metal feature enabled: {:?}", device)
+            }
+        }
+    }
+    fn bench_name<S: Into<String>>(&self, name: S) -> String {
+        match self {
+            Device::Cpu => {
+                let cpu_type = if cfg!(feature = "accelerate") {
+                    "accelerate"
+                } else if cfg!(feature = "mkl") {
+                    "mkl"
+                } else {
+                    "cpu"
+                };
+                format!("{}_{}", cpu_type, name.into())
+            }
+            Device::Cuda(_) => format!("cuda_{}", name.into()),
+            Device::Metal(_) => format!("metal_{}", name.into()),
+        }
+    }
+}
+struct BenchDeviceHandler {
+    devices: Vec<Device>,
+}
+impl BenchDeviceHandler {
+    pub fn new() -> Result<Self> {
+        let mut devices = Vec::new();
+        if cfg!(feature = "metal") {
+            devices.push(Device::new_metal(0)?);
+        } else if cfg!(feature = "cuda") {
+            devices.push(Device::new_cuda(0)?);
+        }
+        devices.push(Device::Cpu);
+        Ok(Self { devices })
+    }
+}
--- a/candle-core/benches/benchmarks/random.rs
+++ b/candle-core/benches/benchmarks/random.rs
+use crate::benchmarks::{BenchDevice, BenchDeviceHandler};
+use candle_core::{DType, Device, Tensor};
+use criterion::{black_box, criterion_group, Criterion, Throughput};
+use std::time::Instant;
+fn rand_uniform(a: &Tensor) {
+    a.rand_like(-1.0, 123.0).unwrap();
+}
+fn rand_normal(a: &Tensor) {
+    a.randn_like(100.0, 15.0).unwrap();
+}
+fn run_random_bench(c: &mut Criterion, device: &Device) {
+    let b = 1;
+    let rows = 2048;
+    let cols = 2048;
+    let dtype = DType::F32;
+    let tensor = Tensor::zeros((b, rows, cols), dtype, device).unwrap();
+    let flops = b * rows * cols * dtype.size_in_bytes();
+    let mut group = c.benchmark_group(device.bench_name("random_uniform"));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |benches| {
+        benches.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                rand_uniform(black_box(&tensor));
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+    let tensor = Tensor::zeros((b, rows, cols), dtype, device).unwrap();
+    let mut group = c.benchmark_group(device.bench_name("random_normal"));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |benches| {
+        benches.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                rand_normal(black_box(&tensor));
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+}
+fn criterion_benchmark(c: &mut Criterion) {
+    let handler = BenchDeviceHandler::new().unwrap();
+    for device in handler.devices {
+        run_random_bench(c, &device);
+    }
+}
+criterion_group!(benches, criterion_benchmark);
--- a/candle-core/benches/benchmarks/where_cond.rs
+++ b/candle-core/benches/benchmarks/where_cond.rs
+use crate::benchmarks::{BenchDevice, BenchDeviceHandler};
+use candle_core::{DType, Device, Tensor};
+use criterion::{black_box, criterion_group, Criterion, Throughput};
+use std::time::Instant;
+fn run(a: &Tensor, b: &Tensor, c: &Tensor) {
+    a.where_cond(b, c).unwrap();
+}
+const fn create_cond_arr<const N: usize>() -> [u8; N] {
+    let mut arr = [0u8; N];
+    let mut i = 0;
+    while i < N {
+        arr[i] = (i % 2) as u8;
+        i += 1;
+    }
+    arr
+}
+const B: usize = 1;
+const M: usize = 1024;
+const K: usize = 1024;
+const SIZE: usize = B * M * K;
+const DATA: [u8; SIZE] = create_cond_arr::<SIZE>();
+fn run_where_cond_benchmark(c: &mut Criterion, device: &Device, dtype: DType, name: &str) {
+    let tensor = Tensor::from_slice(DATA.as_slice(), (B, M, K), &device).unwrap();
+    let on_true = Tensor::ones((B, M, K), dtype, &device).unwrap();
+    let on_false = Tensor::zeros((B, M, K), dtype, &device).unwrap();
+    let elements = B * M * K;
+    // E.g. 2 f32 tensors + 1 u8 tensor
+    let flops = (2 * elements * dtype.size_in_bytes()) + elements;
+    let mut group = c.benchmark_group(device.bench_name(name));
+    group.throughput(Throughput::Bytes(flops as u64));
+    group.bench_function("iter", move |b| {
+        b.iter_custom(|iters| {
+            let start = Instant::now();
+            for _i in 0..iters {
+                run(
+                    black_box(&tensor),
+                    black_box(&on_true),
+                    black_box(&on_false),
+                );
+            }
+            device.sync().unwrap();
+            start.elapsed()
+        })
+    });
+    group.finish();
+}
+fn criterion_benchmark(c: &mut Criterion) {
+    let device = BenchDeviceHandler::new().unwrap();
+    for d in device.devices {
+        run_where_cond_benchmark(c, &d, DType::F32, "where_cond_f32");
+        run_where_cond_benchmark(c, &d, DType::BF16, "where_cond_bf16");
+        run_where_cond_benchmark(c, &d, DType::F16, "where_cond_f16");
+    }
+}
+criterion_group!(benches, criterion_benchmark);
--- a/candle-core/examples/basics.rs
+++ b/candle-core/examples/basics.rs
+#[cfg(any(feature = "mkl", feature = "mkl-dynamic"))]
+extern crate intel_mkl_src;
+#[cfg(feature = "accelerate")]
+extern crate accelerate_src;
+use anyhow::Result;
+use candle_core::{Device, Tensor};
+fn main() -> Result<()> {
+    let a = Tensor::new(&[[0.0f32, 1.0, 2.0], [3.0, 4.0, 5.0]], &Device::Cpu)?;
+    let b = Tensor::new(&[[88.0f32, 99.0]], &Device::Cpu)?;
+    let new_a = a.slice_scatter(&b, 1, 2)?;
+    assert_eq!(a.to_vec2::<f32>()?, [[0.0, 1.0, 2.0], [3.0, 4.0, 5.0]]);
+    assert_eq!(new_a.to_vec2::<f32>()?, [[0.0, 1.0, 2.0], [3.0, 4.0, 5.0]]);
+    Ok(())
+}
--- a/candle-core/examples/cuda_basics.rs
+++ b/candle-core/examples/cuda_basics.rs
+#[cfg(feature = "accelerate")]
+extern crate accelerate_src;
+#[cfg(any(feature = "mkl", feature = "mkl-dynamic"))]
+extern crate intel_mkl_src;
+use anyhow::Result;
+use candle_core::{Device, Module, Tensor};
+use candle_core::quantized::{QMatMul, QTensor};
+fn main() -> Result<()> {
+    let device = Device::new_cuda(0)?;
+    let q = Tensor::randn(0f32, 1.0, (72, 256), &device)?;
+    let q_cpu = q.to_device(&Device::Cpu)?;
+    let q = QTensor::quantize(&q, candle_core::quantized::GgmlDType::Q8K)?;
+    let q = QMatMul::from_qtensor(q)?;
+    let x = Tensor::randn(0f32, 1.0, (5, 256), &device)?;
+    let res_q_cuda = q.forward(&x)?;
+    println!("{res_q_cuda}");
+    let q_cpu = QTensor::quantize(&q_cpu, candle_core::quantized::GgmlDType::Q8K)?;
+    let q_cpu_tensor = q_cpu.dequantize(&Device::Cpu)?;
+    let q_cpu = QMatMul::from_qtensor(q_cpu)?;
+    let x_cpu = x.to_device(&Device::Cpu)?;
+    let res_q_cpu = q_cpu.forward(&x_cpu)?;
+    println!("{res_q_cpu}");
+    let res_mm = x_cpu.matmul(&q_cpu_tensor.t()?)?;
+    let diff = (res_mm - res_q_cuda.to_device(&Device::Cpu))?
+        .abs()?
+        .flatten_all()?
+        .max(0)?;
+    println!("{diff}");
+    Ok(())
+}