Repository navigation
Expand file tree
/
Copy pathmodels.json
More file actions
1 lines (1 loc) · 19.1 KB
/
Copy pathmodels.json
File metadata and controls
1 lines (1 loc) · 19.1 KB
1
{"bge-small-en-v1.5":{"name":"BGE Small En v1.5","description":"High-performance, lightweight text embedding model by BAAI.","backend":"openvino-embeddings","model_path":"models/openvino/bge-small-en-v1.5","source_model":"BAAI/bge-small-en-v1.5","weight_format":"fp16","recommended_device":"CPU","max_context_len":512,"max_output_tokens":0},"tinyllama-1.1b-chat-fp16":{"name":"TinyLlama 1.1B Chat FP16","description":"Small NPU validation model for OpenVINO GenAI.","backend":"openvino-genai","model_path":"models/openvino/tinyllama-1.1b-chat-fp16","source_model":"TinyLlama/TinyLlama-1.1B-Chat-v1.0","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"tinyllama-1.1b-chat-int4":{"name":"TinyLlama 1.1B Chat INT4","description":"Compact INT4 starter model for lower storage and memory pressure.","backend":"openvino-genai","model_path":"models/openvino/tinyllama-1.1b-chat-int4","source_model":"TinyLlama/TinyLlama-1.1B-Chat-v1.0","weight_format":"int4","recommended_device":"CPU","max_context_len":2048,"max_output_tokens":512},"qwen2.5-1.5b-fp16":{"name":"Qwen2.5 1.5B Instruct FP16","description":"Compact Qwen instruct model exported in the FP16 format used by Intel NPU.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-1.5b-instruct-fp16","source_model":"Qwen/Qwen2.5-1.5B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"qwen2.5-3b-fp16":{"name":"Qwen2.5 3B Instruct FP16","description":"Higher-quality Qwen instruct model with retained CPU and GPU certification. Candidate for manual NPU testing.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-3b-instruct-fp16","source_model":"Qwen/Qwen2.5-3B-Instruct","weight_format":"fp16","recommended_device":"GPU","max_context_len":4096,"max_output_tokens":1024},"phi-3.5-mini-fp16":{"name":"Phi-3.5 Mini Instruct FP16","description":"Microsoft Phi-3.5 mini exported as FP16 for Intel NPU testing.","backend":"openvino-genai","model_path":"models/openvino/phi-3.5-mini-instruct-fp16","source_model":"microsoft/Phi-3.5-mini-instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.2-3b-fp16":{"name":"Llama 3.2 3B Instruct FP16","description":"Meta Llama 3.2 3B exported as FP16 for NPU. Publisher approval is required on Hugging Face.","backend":"openvino-genai","model_path":"models/openvino/llama-3.2-3b-instruct-fp16","source_model":"meta-llama/Llama-3.2-3B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-0.5b-fp16":{"name":"Qwen2.5 0.5B Instruct FP16","description":"Ultra-compact Qwen instruct model for CPU testing and candidate NPU testing.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-0.5b-instruct-fp16","source_model":"Qwen/Qwen2.5-0.5B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"deepseek-r1-distill-qwen-1.5b-fp16":{"name":"DeepSeek-R1 Distill Qwen 1.5B FP16","description":"Reasoning model distilled from DeepSeek-R1, exported as FP16 for NPU.","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-1.5b-fp16","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.2-1b-fp16":{"name":"Llama 3.2 1B Instruct FP16","description":"Meta Llama 3.2 1B exported as FP16 for NPU. Publisher approval is required on Hugging Face.","backend":"openvino-genai","model_path":"models/openvino/llama-3.2-1b-instruct-fp16","source_model":"meta-llama/Llama-3.2-1B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"smollm2-135m-fp16":{"name":"SmolLM2 135M Instruct FP16","description":"Ultra-lightweight SmolLM2 model (135M parameters) for lightning-fast testing and NPU/CPU debugging.","backend":"openvino-genai","model_path":"models/openvino/smollm2-135m-instruct-fp16","source_model":"HuggingFaceTB/SmolLM2-135M-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"smollm2-360m-fp16":{"name":"SmolLM2 360M Instruct FP16","description":"Efficient small model (360M parameters) for compatibility testing; NPU behavior depends on platform and driver.","backend":"openvino-genai","model_path":"models/openvino/smollm2-360m-instruct-fp16","source_model":"HuggingFaceTB/SmolLM2-360M-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"smollm2-1.7b-fp16":{"name":"SmolLM2 1.7B Instruct FP16","description":"Small SmolLM2 FP16 model and candidate for NPU testing; compatibility depends on platform and driver.","backend":"openvino-genai","model_path":"models/openvino/smollm2-1.7b-instruct-fp16","source_model":"HuggingFaceTB/SmolLM2-1.7B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"gemma-2-2b-fp16":{"name":"Gemma 2 2B Instruct FP16","description":"Google Gemma 2 2B Instruct exported as FP16 for NPU. Publisher approval is required on Hugging Face.","backend":"openvino-genai","model_path":"models/openvino/gemma-2-2b-instruct-fp16","source_model":"google/gemma-2-2b-it","access_type":"gated","model_url":"https://huggingface.co/google/gemma-2-2b-it","license_url":"https://huggingface.co/google/gemma-2-2b-it","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"phi-4-mini-fp16":{"name":"Phi-4 Mini Instruct FP16","description":"Microsoft Phi-4 mini reasoning model exported as FP16 for Intel NPU testing.","backend":"openvino-genai","model_path":"models/openvino/phi-4-mini-instruct-fp16","source_model":"microsoft/phi-4-mini-instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-7b-fp16":{"name":"Qwen2.5 7B Instruct FP16","description":"Full-sized Qwen2.5 7B model for high-end systems, exported as FP16.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-7b-instruct-fp16","source_model":"Qwen/Qwen2.5-7B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.1-8b-fp16":{"name":"Llama 3.1 8B Instruct FP16","description":"Meta Llama 3.1 8B exported as FP16. Publisher approval is required on Hugging Face.","backend":"openvino-genai","model_path":"models/openvino/llama-3.1-8b-instruct-fp16","source_model":"meta-llama/Llama-3.1-8B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","weight_format":"fp16","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-14b-fp16":{"name":"Qwen2.5 14B Instruct FP16","description":"Mid-sized Qwen model for high-end systems (requires 64GB RAM).","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-14b-instruct-fp16","source_model":"Qwen/Qwen2.5-14B-Instruct","weight_format":"fp16","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"deepseek-r1-distill-qwen-14b-fp16":{"name":"DeepSeek-R1 Distill Qwen 14B FP16","description":"High-quality 14B reasoning model distilled from DeepSeek-R1 (requires 64GB RAM).","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-14b-fp16","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-14B","weight_format":"fp16","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-32b-fp16":{"name":"Qwen2.5 32B Instruct FP16","description":"Very large Qwen2.5 model for extreme accuracy (requires 64GB RAM).","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-32b-instruct-fp16","source_model":"Qwen/Qwen2.5-32B-Instruct","weight_format":"fp16","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"deepseek-r1-distill-qwen-32b-fp16":{"name":"DeepSeek-R1 Distill Qwen 32B FP16","description":"High-end 32B reasoning model distilled from DeepSeek-R1 (requires 64GB RAM).","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-32b-fp16","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","weight_format":"fp16","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-0.5b-int4":{"name":"Qwen2.5 0.5B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-0.5b-instruct-int4","source_model":"Qwen/Qwen2.5-0.5B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"qwen2.5-1.5b-int4":{"name":"Qwen2.5 1.5B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-1.5b-instruct-int4","source_model":"Qwen/Qwen2.5-1.5B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"qwen2.5-3b-int4":{"name":"Qwen2.5 3B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-3b-instruct-int4","source_model":"Qwen/Qwen2.5-3B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"phi-3.5-mini-int4":{"name":"Phi-3.5 Mini Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/phi-3.5-mini-instruct-int4","source_model":"microsoft/Phi-3.5-mini-instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.2-3b-int4":{"name":"Llama 3.2 3B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/llama-3.2-3b-instruct-int4","source_model":"meta-llama/Llama-3.2-3B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"deepseek-r1-distill-qwen-1.5b-int4":{"name":"DeepSeek-R1 Distill Qwen 1.5B INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-1.5b-int4","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.2-1b-int4":{"name":"Llama 3.2 1B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/llama-3.2-1b-instruct-int4","source_model":"meta-llama/Llama-3.2-1B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"smollm2-135m-int4":{"name":"SmolLM2 135M Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/smollm2-135m-instruct-int4","source_model":"HuggingFaceTB/SmolLM2-135M-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"smollm2-360m-int4":{"name":"SmolLM2 360M Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/smollm2-360m-instruct-int4","source_model":"HuggingFaceTB/SmolLM2-360M-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"smollm2-1.7b-int4":{"name":"SmolLM2 1.7B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/smollm2-1.7b-instruct-int4","source_model":"HuggingFaceTB/SmolLM2-1.7B-Instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":2048,"max_output_tokens":512},"gemma-2-2b-int4":{"name":"Gemma 2 2B Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/gemma-2-2b-instruct-int4","source_model":"google/gemma-2-2b-it","access_type":"gated","model_url":"https://huggingface.co/google/gemma-2-2b-it","license_url":"https://huggingface.co/google/gemma-2-2b-it","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"phi-4-mini-int4":{"name":"Phi-4 Mini Instruct INT4","description":"Compressed INT4 candidate using InferBridge's NPU-safe symmetric conversion profile. Validate on local hardware before treating it as certified.","backend":"openvino-genai","model_path":"models/openvino/phi-4-mini-instruct-int4","source_model":"microsoft/phi-4-mini-instruct","weight_format":"int4","recommended_device":"NPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-7b-int4":{"name":"Qwen2.5 7B Instruct INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-7b-instruct-int4","source_model":"Qwen/Qwen2.5-7B-Instruct","weight_format":"int4","recommended_device":"GPU","max_context_len":4096,"max_output_tokens":1024},"llama-3.1-8b-int4":{"name":"Llama 3.1 8B Instruct INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/llama-3.1-8b-instruct-int4","source_model":"meta-llama/Llama-3.1-8B-Instruct","access_type":"gated","model_url":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","license_url":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","weight_format":"int4","recommended_device":"GPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-14b-int4":{"name":"Qwen2.5 14B Instruct INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-14b-instruct-int4","source_model":"Qwen/Qwen2.5-14B-Instruct","weight_format":"int4","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"deepseek-r1-distill-qwen-14b-int4":{"name":"DeepSeek-R1 Distill Qwen 14B INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-14b-int4","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-14B","weight_format":"int4","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-32b-int4":{"name":"Qwen2.5 32B Instruct INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-32b-instruct-int4","source_model":"Qwen/Qwen2.5-32B-Instruct","weight_format":"int4","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"deepseek-r1-distill-qwen-32b-int4":{"name":"DeepSeek-R1 Distill Qwen 32B INT4","description":"Compressed INT4 candidate for substantially lower model storage and runtime-memory pressure. Benchmark this variant on the target hardware before promoting it over FP16.","backend":"openvino-genai","model_path":"models/openvino/deepseek-r1-distill-qwen-32b-int4","source_model":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","weight_format":"int4","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-1.5b-int8":{"name":"Qwen2.5 1.5B Instruct INT8","description":"INT8 quality-focused compressed candidate for CPU/GPU comparison against INT4 and the FP16 fallback. Benchmark locally before making performance or quality claims.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-1.5b-instruct-int8","source_model":"Qwen/Qwen2.5-1.5B-Instruct","weight_format":"int8","recommended_device":"GPU","max_context_len":2048,"max_output_tokens":512},"qwen2.5-3b-int8":{"name":"Qwen2.5 3B Instruct INT8","description":"INT8 quality-focused compressed candidate for CPU/GPU comparison against INT4 and the FP16 fallback. Benchmark locally before making performance or quality claims.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-3b-instruct-int8","source_model":"Qwen/Qwen2.5-3B-Instruct","weight_format":"int8","recommended_device":"GPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-7b-int8":{"name":"Qwen2.5 7B Instruct INT8","description":"INT8 quality-focused compressed candidate for CPU/GPU comparison against INT4 and the FP16 fallback. Benchmark locally before making performance or quality claims.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-7b-instruct-int8","source_model":"Qwen/Qwen2.5-7B-Instruct","weight_format":"int8","recommended_device":"GPU","max_context_len":4096,"max_output_tokens":1024},"qwen2.5-14b-int8":{"name":"Qwen2.5 14B Instruct INT8","description":"INT8 quality-focused compressed candidate for CPU/GPU comparison against INT4 and the FP16 fallback. Benchmark locally before making performance or quality claims.","backend":"openvino-genai","model_path":"models/openvino/qwen2.5-14b-instruct-int8","source_model":"Qwen/Qwen2.5-14B-Instruct","weight_format":"int8","recommended_device":"CPU","max_context_len":4096,"max_output_tokens":1024}}