Metal quantized modifications proposal.

- Add a device param, wherever needed. - Create new QMetal storage thing that implements QuantizedType. - Update everywhere needed. Fix Python. Fixing examples. Fix: fmt + clippy + stub. Moving everything around. Only missing the actual implems. Fixing everything + adding dequantized kernels. More work. Fixing matmul. Fmt + Clippy Some clippy fixes. Working state. Q2K Metal -> Bugged (also present in GGML). Q4K CPU -> Bugged (present previously, new test catch it). Q5K CPU -> Bugged (present previously). Q8_1 Both -> Never really implemented it seems Q8K metal -> Never implemented in metal Fixing Q2K bug (present in ggml).
2025-06-21 12:20:46 +00:00 · 2023-11-18 14:49:30 +01:00
parent ea36f3b11f
commit 9c4b4f0da0
31 changed files with 6400 additions and 419 deletions
--- a/candle-examples/examples/phi/main.rs
+++ b/candle-examples/examples/phi/main.rs
@ -307,16 +307,21 @@ fn main() -> Result<()> {
        WhichModel::PuffinPhiV2 => Config::puffin_phi_v2(),
        WhichModel::PhiHermes => Config::phi_hermes_1_3b(),
    };
-    let (model, device) = if args.quantized {
-        let vb = candle_transformers::quantized_var_builder::VarBuilder::from_gguf(&filenames[0])?;
+    let device = candle_examples::device(args.cpu)?;
+    let model = if args.quantized {
        let config = config();
+        let vb = candle_transformers::quantized_var_builder::VarBuilder::from_gguf(
+            &filenames[0],
+            &device,
+        )?;
+        println!("Loaded vb");
        let model = match args.model {
            WhichModel::V2 | WhichModel::V2Old => QMixFormer::new_v2(&config, vb)?,
            _ => QMixFormer::new(&config, vb)?,
        };
-        (Model::Quantized(model), Device::Cpu)
+        println!("Loaded model");
+        Model::Quantized(model)
    } else {
-        let device = candle_examples::device(args.cpu)?;
        let vb = unsafe { VarBuilder::from_mmaped_safetensors(&filenames, DType::F32, &device)? };
        let model = match args.model {
            WhichModel::V1 | WhichModel::V1_5 | WhichModel::V2 => {
@ -335,7 +340,7 @@ fn main() -> Result<()> {
                Model::MixFormer(MixFormer::new(&config, vb)?)
            }
        };
-        (model, device)
+        model
    };
    println!("loaded the model in {:?}", start.elapsed());