abetlen
diff --git a/‎.github/workflows/build-and-release.yaml
Lines changed: 2 additions & 2 deletions b/‎.github/workflows/build-and-release.yaml
Lines changed: 2 additions & 2 deletions
diff --git a/‎.github/workflows/build-wheels-metal.yaml
Lines changed: 1 addition & 1 deletion b/‎.github/workflows/build-wheels-metal.yaml
Lines changed: 1 addition & 1 deletion
diff --git a/‎.github/workflows/generate-index-from-release.yaml
Lines changed: 8 additions & 3 deletions b/‎.github/workflows/generate-index-from-release.yaml
Lines changed: 8 additions & 3 deletions
diff --git a/‎CHANGELOG.md
Lines changed: 46 additions & 0 deletions b/‎CHANGELOG.md
Lines changed: 46 additions & 0 deletions
diff --git a/‎README.md
Lines changed: 6 additions & 1 deletion b/‎README.md
Lines changed: 6 additions & 1 deletion
diff --git a/‎docker/cuda_simple/Dockerfile
Lines changed: 2 additions & 2 deletions b/‎docker/cuda_simple/Dockerfile
Lines changed: 2 additions & 2 deletions
diff --git a/‎docker/open_llama/Dockerfile
Lines changed: 3 additions & 3 deletions b/‎docker/open_llama/Dockerfile
Lines changed: 3 additions & 3 deletions
diff --git a/‎docker/openblas_simple/Dockerfile
Lines changed: 1 addition & 1 deletion b/‎docker/openblas_simple/Dockerfile
Lines changed: 1 addition & 1 deletion
diff --git a/‎docker/simple/Dockerfile
Lines changed: 1 addition & 1 deletion b/‎docker/simple/Dockerfile
Lines changed: 1 addition & 1 deletion
diff --git a/‎docs/icon.svg
Lines changed: 5 additions & 0 deletions b/‎docs/icon.svg
Lines changed: 5 additions & 0 deletions
diff --git a/‎llama_cpp/__init__.py
Lines changed: 1 addition & 1 deletion b/‎llama_cpp/__init__.py
Lines changed: 1 addition & 1 deletion
diff --git a/‎llama_cpp/_internals.py
Lines changed: 5 additions & 5 deletions b/‎llama_cpp/_internals.py
Lines changed: 5 additions & 5 deletions
diff --git a/‎llama_cpp/llama.py
Lines changed: 25 additions & 14 deletions b/‎llama_cpp/llama.py
Lines changed: 25 additions & 14 deletions
@@ -42,7 +42,7 @@ jobs:
         shell: cmd
 
       - name: Build wheels
-        uses: pypa/cibuildwheel@v2.19.2
+        uses: pypa/cibuildwheel@v2.20.0
         env:
           # disable repair
           CIBW_REPAIR_WHEEL_COMMAND: ""
@@ -69,7 +69,7 @@ jobs:
           platforms: linux/arm64
 
       - name: Build wheels
-        uses: pypa/cibuildwheel@v2.19.2
+        uses: pypa/cibuildwheel@v2.20.0
         env:
           CIBW_SKIP: "*musllinux* pp*"
           CIBW_REPAIR_WHEEL_COMMAND: ""
 
@@ -43,7 +43,7 @@ jobs:
         shell: cmd
 
       - name: Build wheels
-        uses: pypa/cibuildwheel@v2.19.2
+        uses: pypa/cibuildwheel@v2.20.0
         env:
           # disable repair
           CIBW_REPAIR_WHEEL_COMMAND: ""
 
@@ -1,9 +1,11 @@
 name: Wheels Index
 
 on:
-  # Trigger on any new release
-  release:
-    types: [published]
+  # Trigger on new release
+  workflow_run:
+    workflows: ["Release", "Build Wheels (CUDA)", "Build Wheels (Metal)"]
+    types:
+      - completed
 
   # Allows you to run this workflow manually from the Actions tab
   workflow_dispatch:
@@ -33,7 +35,10 @@ jobs:
       - name: Setup Pages
         uses: actions/configure-pages@v5
       - name: Build
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
         run: |
+          ./scripts/get-releases.sh
           ./scripts/releases-to-pep-503.sh index/whl/cpu '^[v]?[0-9]+\.[0-9]+\.[0-9]+$'
           ./scripts/releases-to-pep-503.sh index/whl/cu121 '^[v]?[0-9]+\.[0-9]+\.[0-9]+-cu121$'
           ./scripts/releases-to-pep-503.sh index/whl/cu122 '^[v]?[0-9]+\.[0-9]+\.[0-9]+-cu122$'
 
@@ -7,6 +7,52 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
 
 ## [Unreleased]
 
+## [0.2.90]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@1d1ccce67613674c75c9c7e3fa4c1e24e428ba48
+- feat: Add support for `MiniCPMv26ChatHandler` and `minicpm-v-26` in server by @abetlen in f70df824985d875226793b94dacc0c302a4256b2
+
+## [0.2.89]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@cfac111e2b3953cdb6b0126e67a2487687646971
+- fix: Llama.close didn't free lora adapter by @jkawamoto in #1679
+- fix: missing dependencies for test by @jkawamoto in #1680
+
+## [0.2.88]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@fc4ca27b25464a11b3b86c9dbb5b6ed6065965c2
+- fix: only print 'cache saved' in verbose mode by @lsorber in #1668 
+- fix: Added back from_file method to LlamaGrammar by @ExtReMLapin in #1673
+- fix: grammar prints on each call by @abetlen in 0998ea0deea076a547d54bd598d6b413b588ee2b
+- feat: Enable recursive search of HFFS.ls when using from_pretrained by @benHeidabetlen in #1656
+- feat: Add more detailed log for prefix-match by @xu-song in #1659
+
+## [0.2.87]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@be55695eff44784a141a863f273661a6bce63dfc
+- fix: Include all llama.cpp source files and subdirectories by @abetlen in 9cad5714ae6e7c250af8d0bbb179f631368c928b
+- feat(ci): Re-build wheel index automatically when releases are created by @abetlen in 198f47dc1bd202fd2b71b29e041a9f33fe40bfad
+
+## [0.2.86]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@398ede5efeb07b9adf9fbda7ea63f630d476a792
+- feat: Ported back new grammar changes from C++ to Python implementation by @ExtReMLapin in (#1637)
+- fix: llama_grammar_accept_token arg order by @tc-wolf in (#1649)
+
+## [0.2.85]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@398ede5efeb07b9adf9fbda7ea63f630d476a792
+- fix: Missing LoRA adapter after API change by @shamitv in #1630
+- fix(docker): Update Dockerfile BLAS options by @olivierdebauche in #1632
+- fix(docker): Fix GGML_CUDA param by @olivierdebauche in #1633
+- fix(docker): Update Dockerfile build options from `LLAMA_` to `GGML_` by @olivierdebauche in #1634
+- feat: FreeBSD compatibility by @yurivict in #1635
+
+## [0.2.84]
+
+- feat: Update llama.cpp to ggerganov/llama.cpp@4730faca618ff9cee0780580145e3cbe86f24876
+- fix: fix: Correcting run.sh filepath in Simple Docker implementation by @mashuk999 in #1626
+
 ## [0.2.83]
 
 - feat: Update llama.cpp to ggerganov/llama.cpp@081fe431aa8fb6307145c4feb3eed4f48cab19f8
 
@@ -1,4 +1,8 @@
-# 🦙 Python Bindings for [`llama.cpp`](https://github.com/ggerganov/llama.cpp)
+<p align="center">
+  <img src="https://raw.githubusercontent.com/abetlen/llama-cpp-python/main/docs/icon.svg" style="height: 5rem; width: 5rem">
+</p>
+
+#  Python Bindings for [`llama.cpp`](https://github.com/ggerganov/llama.cpp)
 
 [![Documentation Status](https://readthedocs.org/projects/llama-cpp-python/badge/?version=latest)](https://llama-cpp-python.readthedocs.io/en/latest/?badge=latest)
 [![Tests](https://github.com/abetlen/llama-cpp-python/actions/workflows/test.yaml/badge.svg?branch=main)](https://github.com/abetlen/llama-cpp-python/actions/workflows/test.yaml)
@@ -500,6 +504,7 @@ Below are the supported multi-modal models and their respective chat handlers (P
 | [moondream2](https://huggingface.co/vikhyatk/moondream2) | `MoondreamChatHandler` | `moondream2` |
 | [nanollava](https://huggingface.co/abetlen/nanollava-gguf) | `NanollavaChatHandler` | `nanollava` |
 | [llama-3-vision-alpha](https://huggingface.co/abetlen/llama-3-vision-alpha-gguf) | `Llama3VisionAlphaChatHandler` | `llama-3-vision-alpha` |
+| [minicpm-v-2.6](https://huggingface.co/openbmb/MiniCPM-V-2_6-gguf) | `MiniCPMv26ChatHandler` | `minicpm-v-2.6` |
 
 Then you'll need to use a custom chat handler to load the clip model and process the chat messages and images.
 
 
@@ -15,13 +15,13 @@ COPY . .
 
 # setting build related env vars
 ENV CUDA_DOCKER_ARCH=all
-ENV LLAMA_CUBLAS=1
+ENV GGML_CUDA=1
 
 # Install depencencies
 RUN python3 -m pip install --upgrade pip pytest cmake scikit-build setuptools fastapi uvicorn sse-starlette pydantic-settings starlette-context
 
 # Install llama-cpp-python (build with cuda)
-RUN CMAKE_ARGS="-DLLAMA_CUBLAS=on" pip install llama-cpp-python
+RUN CMAKE_ARGS="-DGGML_CUDA=on" pip install llama-cpp-python
 
 # Run the server
 CMD python3 -m llama_cpp.server
@@ -20,13 +20,13 @@ RUN python3 -m pip install --upgrade pip pytest cmake scikit-build setuptools fa
 
 # Perform the conditional installations based on the image
 RUN echo "Image: ${IMAGE}" && \
-    if [ "${IMAGE}" = "python:3-slim-bullseye" ] ; then \
+    if [ "${IMAGE}" = "python:3-slim-bookworm" ] ; then \
     echo "OpenBLAS install:" && \
     apt-get install -y --no-install-recommends libopenblas-dev && \
-    LLAMA_OPENBLAS=1 pip install llama-cpp-python --verbose; \
+    CMAKE_ARGS="-DGGML_BLAS=ON -DGGML_BLAS_VENDOR=OpenBLAS" pip install llama-cpp-python --verbose; \
 else \
     echo "CuBLAS install:" && \
-    LLAMA_CUBLAS=1 pip install llama-cpp-python --verbose; \
+    CMAKE_ARGS="-DGGML_CUDA=on" pip install llama-cpp-python --verbose; \
 fi
 
 # Clean up apt cache
 
@@ -12,7 +12,7 @@ RUN apt update && apt install -y libopenblas-dev ninja-build build-essential pkg
 
 RUN python -m pip install --upgrade pip pytest cmake scikit-build setuptools fastapi uvicorn sse-starlette pydantic-settings starlette-context
 
-RUN CMAKE_ARGS="-DLLAMA_BLAS=ON -DLLAMA_BLAS_VENDOR=OpenBLAS" pip install llama_cpp_python --verbose
+RUN CMAKE_ARGS="-DGGML_BLAS=ON -DGGML_BLAS_VENDOR=OpenBLAS" pip install llama_cpp_python --verbose
 
 # Run the server
 CMD python3 -m llama_cpp.server
@@ -35,4 +35,4 @@ ENV PORT=8000
 EXPOSE 8000
 
 # Run the server start script
-CMD ["/bin/sh", "/app/run.sh"]
+CMD ["/bin/sh", "/app/docker/simple/run.sh"]
@@ -1,4 +1,4 @@
 from .llama_cpp import *
 from .llama import *
 
-__version__ = "0.2.83"
+__version__ = "0.2.90"
@@ -179,11 +179,11 @@ def token_eot(self) -> int:
         assert self.model is not None
         return llama_cpp.llama_token_eot(self.model)
 
-    def add_bos_token(self) -> int:
+    def add_bos_token(self) -> bool:
         assert self.model is not None
         return llama_cpp.llama_add_bos_token(self.model)
 
-    def add_eos_token(self) -> int:
+    def add_eos_token(self) -> bool:
         assert self.model is not None
         return llama_cpp.llama_add_eos_token(self.model)
 
@@ -511,7 +511,7 @@ def sample_token(self, candidates: "_LlamaTokenDataArray") -> int:
     def grammar_accept_token(self, grammar: LlamaGrammar, token: int):
         assert self.ctx is not None
         assert grammar.grammar is not None
-        llama_cpp.llama_grammar_accept_token(self.ctx, grammar.grammar, token)
+        llama_cpp.llama_grammar_accept_token(grammar.grammar, self.ctx, token)
 
     def reset_timings(self):
         assert self.ctx is not None
@@ -691,8 +691,8 @@ def _detokenize_bpe(model: _LlamaModel, tokens: List[int]) -> str:
 def _should_add_bos(model: _LlamaModel) -> bool:
     assert model.model is not None
     add_bos = llama_cpp.llama_add_bos_token(model.model)
-    if add_bos != -1:
-        return add_bos != 0
+    if add_bos:
+        return add_bos
     else:
         return llama_cpp.llama_vocab_type(model.model) == llama_cpp.LLAMA_VOCAB_TYPE_SPM
 
 
@@ -153,7 +153,7 @@ def __init__(
             model_path: Path to the model.
             n_gpu_layers: Number of layers to offload to GPU (-ngl). If -1, all layers are offloaded.
             split_mode: How to split the model across GPUs. See llama_cpp.LLAMA_SPLIT_* for options.
-            main_gpu: main_gpu interpretation depends on split_mode: LLAMA_SPLIT_NONE: the GPU that is used for the entire model. LLAMA_SPLIT_ROW: the GPU that is used for small tensors and intermediate results. LLAMA_SPLIT_LAYER: ignored
+            main_gpu: main_gpu interpretation depends on split_mode: LLAMA_SPLIT_MODE_NONE: the GPU that is used for the entire model. LLAMA_SPLIT_MODE_ROW: the GPU that is used for small tensors and intermediate results. LLAMA_SPLIT_MODE_LAYER: ignored
             tensor_split: How split tensors should be distributed across GPUs. If None, the model is not split.
             rpc_servers: Comma separated list of RPC servers to use for offloading
             vocab_only: Only load the vocabulary no weights.
@@ -198,6 +198,7 @@ def __init__(
             A Llama instance.
         """
         self.verbose = verbose
+        self._stack = contextlib.ExitStack()
 
         set_verbose(verbose)
 
@@ -365,8 +366,6 @@ def __init__(
         if not os.path.exists(model_path):
             raise ValueError(f"Model path does not exist: {model_path}")
 
-        self._stack = contextlib.ExitStack()
-
         self._model = self._stack.enter_context(
             contextlib.closing(
                 _LlamaModel(
@@ -420,6 +419,15 @@ def __init__(
                 raise RuntimeError(
                     f"Failed to initialize LoRA adapter from lora path: {self.lora_path}"
                 )
+
+            def free_lora_adapter():
+                if self._lora_adapter is None:
+                    return
+                llama_cpp.llama_lora_adapter_free(self._lora_adapter)
+                self._lora_adapter = None
+
+            self._stack.callback(free_lora_adapter)
+
             assert self._ctx.ctx is not None
             if llama_cpp.llama_lora_adapter_set(
                 self._ctx.ctx, self._lora_adapter, self.lora_scale
@@ -570,6 +578,8 @@ def tokenize(
 
         Args:
             text: The utf-8 encoded string to tokenize.
+            add_bos: Whether to add a beginning of sequence token.
+            special: Whether to tokenize special tokens.
 
         Raises:
             RuntimeError: If the tokenization failed.
@@ -580,18 +590,19 @@ def tokenize(
         return self.tokenizer_.tokenize(text, add_bos, special)
 
     def detokenize(
-        self, tokens: List[int], prev_tokens: Optional[List[int]] = None
+        self, tokens: List[int], prev_tokens: Optional[List[int]] = None, special: bool = False
     ) -> bytes:
         """Detokenize a list of tokens.
 
         Args:
             tokens: The list of tokens to detokenize.
-            prev_tokens: The list of previous tokens. Offset mapping will be performed if provided
+            prev_tokens: The list of previous tokens. Offset mapping will be performed if provided.
+            special: Whether to detokenize special tokens.
 
         Returns:
             The detokenized string.
         """
-        return self.tokenizer_.detokenize(tokens, prev_tokens=prev_tokens)
+        return self.tokenizer_.detokenize(tokens, prev_tokens=prev_tokens, special=special)
 
     def set_cache(self, cache: Optional[BaseLlamaCache]):
         """Set the cache.
@@ -777,11 +788,12 @@ def generate(
                 else:
                     break
             if longest_prefix > 0:
-                if self.verbose:
-                    print("Llama.generate: prefix-match hit", file=sys.stderr)
                 reset = False
                 tokens = tokens[longest_prefix:]
                 self.n_tokens = longest_prefix
+                if self.verbose:
+                    print(f"Llama.generate: {longest_prefix} prefix-match hit, "
+                          f"remaining {len(tokens)} prompt tokens to eval", file=sys.stderr)                    
 
         # Reset the model state
         if reset:
@@ -1057,13 +1069,13 @@ def _create_completion(
 
         if (
             (isinstance(prompt, list) and suffix is None)
-            or self._model.add_bos_token() == 0
+            or not self._model.add_bos_token()
             or bos_tokens[:1] == [-1]
         ):
             bos_tokens = []
 
         if (isinstance(prompt, list) and suffix is None) or (
-            self._model.add_eos_token() != 1 and sep_token_id == -1
+            not self._model.add_eos_token() and sep_token_id == -1
         ):
             eos_tokens = []
 
@@ -1522,7 +1534,8 @@ def logit_bias_processor(
                 if self.verbose:
                     print("Llama._create_completion: cache save", file=sys.stderr)
                 self.cache[prompt_tokens + completion_tokens] = self.save_state()
-                print("Llama._create_completion: cache saved", file=sys.stderr)
+                if self.verbose:
+                    print("Llama._create_completion: cache saved", file=sys.stderr)
             return
 
         if self.cache:
@@ -2086,8 +2099,6 @@ def close(self) -> None:
         self._stack.close()
 
     def __del__(self) -> None:
-        if self._lora_adapter is not None:
-            llama_cpp.llama_lora_adapter_free(self._lora_adapter)
         self.close()
 
     @staticmethod
@@ -2156,7 +2167,7 @@ def from_pretrained(
 
         files = [
             file["name"] if isinstance(file, dict) else file
-            for file in hffs.ls(repo_id)
+            for file in hffs.ls(repo_id, recursive=True)
         ]
 
         # split each file into repo_id, subfolder, filename