|
27 | 27 | - filename: llama-cpp/models/gemma-4-26B-A4B-it-APEX-GGUF/gemma-4-26B-A4B-APEX-Quality.gguf |
28 | 28 | uri: https://huggingface.co/mudler/gemma-4-26B-A4B-it-APEX-GGUF/resolve/main/gemma-4-26B-A4B-APEX-Quality.gguf |
29 | 29 | sha256: a6591d7b41978e6f465acd9d03e96286f70912402c695158fb267ccbfbb740ed |
| 30 | +- &gemma4 |
| 31 | + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" |
| 32 | + name: "gemma-4-26b-a4b-it" |
| 33 | + icon: https://ai.google.dev/static/gemma/images/gemma3.png |
| 34 | + license: gemma |
| 35 | + urls: |
| 36 | + - https://huggingface.co/google/gemma-4-26B-A4B-it |
| 37 | + - https://huggingface.co/ggml-org/gemma-4-26B-A4B-it-GGUF |
| 38 | + description: | |
| 39 | + Google Gemma 4 26B-A4B-IT is an open-source multimodal Mixture-of-Experts model with 26B total parameters and 4B active parameters. It handles text and image input, generating text output, with a 256K context window and support for 140+ languages. The MoE architecture provides strong performance with efficient inference. Well-suited for question answering, summarization, reasoning, and image understanding tasks. |
| 40 | + tags: |
| 41 | + - llm |
| 42 | + - gguf |
| 43 | + - gpu |
| 44 | + - cpu |
| 45 | + - gemma |
| 46 | + - gemma4 |
| 47 | + - gemma-4 |
| 48 | + - multimodal |
| 49 | + overrides: |
| 50 | + backend: llama-cpp |
| 51 | + function: |
| 52 | + automatic_tool_parsing_fallback: true |
| 53 | + grammar: |
| 54 | + disable: true |
| 55 | + known_usecases: |
| 56 | + - chat |
| 57 | + mmproj: mmproj-gemma-4-26B-A4B-it-f16.gguf |
| 58 | + options: |
| 59 | + - use_jinja:true |
| 60 | + parameters: |
| 61 | + model: gemma-4-26B-A4B-it-Q4_K_M.gguf |
| 62 | + template: |
| 63 | + use_tokenizer_template: true |
| 64 | + files: |
| 65 | + - filename: gemma-4-26B-A4B-it-Q4_K_M.gguf |
| 66 | + sha256: 23c6997912cb7fa36147fe05877de73ddbb2a80ff69b18ff171b354dccf2b5b5 |
| 67 | + uri: huggingface://ggml-org/gemma-4-26B-A4B-it-GGUF/gemma-4-26B-A4B-it-Q4_K_M.gguf |
| 68 | + - filename: mmproj-gemma-4-26B-A4B-it-f16.gguf |
| 69 | + sha256: 4107c1c3c299095fbc323f87f4e4cac81dd9527db5ff90808fea669e08244531 |
| 70 | + uri: huggingface://ggml-org/gemma-4-26B-A4B-it-GGUF/mmproj-gemma-4-26B-A4B-it-f16.gguf |
| 71 | +- !!merge <<: *gemma4 |
| 72 | + name: "gemma-4-e2b-it" |
| 73 | + urls: |
| 74 | + - https://huggingface.co/google/gemma-4-E2B-it |
| 75 | + - https://huggingface.co/ggml-org/gemma-4-E2B-it-GGUF |
| 76 | + description: | |
| 77 | + Google Gemma 4 E2B-IT is a lightweight open-source multimodal model with 5B total parameters and 2B effective parameters using selective parameter activation. It handles text and image input, generating text output, with a 256K context window and support for 140+ languages. Optimized for efficient execution on low-resource devices including mobile and laptops. |
| 78 | + overrides: |
| 79 | + backend: llama-cpp |
| 80 | + function: |
| 81 | + automatic_tool_parsing_fallback: true |
| 82 | + grammar: |
| 83 | + disable: true |
| 84 | + known_usecases: |
| 85 | + - chat |
| 86 | + mmproj: mmproj-gemma-4-e2b-it-f16.gguf |
| 87 | + options: |
| 88 | + - use_jinja:true |
| 89 | + parameters: |
| 90 | + model: gemma-4-e2b-it-Q8_0.gguf |
| 91 | + template: |
| 92 | + use_tokenizer_template: true |
| 93 | + files: |
| 94 | + - filename: gemma-4-e2b-it-Q8_0.gguf |
| 95 | + sha256: 12d878964d21f1779dea15abeee048855151b27089fe98b32c628f85740933f3 |
| 96 | + uri: huggingface://ggml-org/gemma-4-E2B-it-GGUF/gemma-4-e2b-it-Q8_0.gguf |
| 97 | + - filename: mmproj-gemma-4-e2b-it-f16.gguf |
| 98 | + sha256: 9165f3d9674c3731ae29373d95b860d141eee030b0ec0bf4577e2de8596a7767 |
| 99 | + uri: huggingface://ggml-org/gemma-4-E2B-it-GGUF/mmproj-gemma-4-e2b-it-f16.gguf |
| 100 | +- !!merge <<: *gemma4 |
| 101 | + name: "gemma-4-e4b-it" |
| 102 | + urls: |
| 103 | + - https://huggingface.co/google/gemma-4-E4B-it |
| 104 | + - https://huggingface.co/ggml-org/gemma-4-E4B-it-GGUF |
| 105 | + description: | |
| 106 | + Google Gemma 4 E4B-IT is an open-source multimodal model with 8B total parameters and 4B effective parameters using selective parameter activation. It handles text and image input, generating text output, with a 256K context window and support for 140+ languages. Offers a good balance of performance and efficiency for deployment on consumer hardware. |
| 107 | + overrides: |
| 108 | + backend: llama-cpp |
| 109 | + function: |
| 110 | + automatic_tool_parsing_fallback: true |
| 111 | + grammar: |
| 112 | + disable: true |
| 113 | + known_usecases: |
| 114 | + - chat |
| 115 | + mmproj: mmproj-gemma-4-e4b-it-f16.gguf |
| 116 | + options: |
| 117 | + - use_jinja:true |
| 118 | + parameters: |
| 119 | + model: gemma-4-e4b-it-Q4_K_M.gguf |
| 120 | + template: |
| 121 | + use_tokenizer_template: true |
| 122 | + files: |
| 123 | + - filename: gemma-4-e4b-it-Q4_K_M.gguf |
| 124 | + sha256: dff4e4ca848e33e678a63b5b7d1f8bfa4a17e764415d0c0aaaad07c84f4d8fad |
| 125 | + uri: huggingface://ggml-org/gemma-4-E4B-it-GGUF/gemma-4-e4b-it-Q4_K_M.gguf |
| 126 | + - filename: mmproj-gemma-4-e4b-it-f16.gguf |
| 127 | + sha256: a7e94f39cee4569fae49c852cbfb574c54e225aafb8b75313a8bf06b89e17712 |
| 128 | + uri: huggingface://ggml-org/gemma-4-E4B-it-GGUF/mmproj-gemma-4-e4b-it-f16.gguf |
| 129 | +- !!merge <<: *gemma4 |
| 130 | + name: "gemma-4-31b-it" |
| 131 | + urls: |
| 132 | + - https://huggingface.co/google/gemma-4-31B-it |
| 133 | + - https://huggingface.co/unsloth/gemma-4-31B-it-GGUF |
| 134 | + description: | |
| 135 | + Google Gemma 4 31B-IT is the largest dense model in the Gemma 4 family with 31B parameters. It handles text and image input, generating text output, with a 256K context window and support for 140+ languages. Provides the highest quality outputs in the Gemma 4 lineup, well-suited for complex reasoning, summarization, and image understanding tasks. |
| 136 | + overrides: |
| 137 | + backend: llama-cpp |
| 138 | + function: |
| 139 | + automatic_tool_parsing_fallback: true |
| 140 | + grammar: |
| 141 | + disable: true |
| 142 | + known_usecases: |
| 143 | + - chat |
| 144 | + mmproj: mmproj-F16.gguf |
| 145 | + options: |
| 146 | + - use_jinja:true |
| 147 | + parameters: |
| 148 | + model: gemma-4-31B-it-Q4_K_M.gguf |
| 149 | + template: |
| 150 | + use_tokenizer_template: true |
| 151 | + files: |
| 152 | + - filename: gemma-4-31B-it-Q4_K_M.gguf |
| 153 | + sha256: 5783acab7217360984ea957abc36f89d35cdeba2f5dd30b0e9e33f9f294bad82 |
| 154 | + uri: huggingface://unsloth/gemma-4-31B-it-GGUF/gemma-4-31B-it-Q4_K_M.gguf |
| 155 | + - filename: mmproj-F16.gguf |
| 156 | + sha256: 1be2a32013e0d29c6c746513b5f5e7d38f47b694351e74a69c7172acdbbb11a6 |
| 157 | + uri: https://huggingface.co/unsloth/gemma-4-31B-it-GGUF/resolve/main/mmproj-F16.gguf |
30 | 158 | - name: "qwen3.5-35b-a3b-apex" |
31 | 159 | url: "github:mudler/LocalAI/gallery/virtual.yaml@master" |
32 | 160 | urls: |
|
0 commit comments