{ "_comment": "Consolidated model information mapping. Contains developer, type, cost information, and execution specifications for each model.", "_units": { "api_costs": "cost per 1M input tokens + cost per 1M output tokens (if applicable)", "local_costs": "cost per 1M tokens (calculated using formula)" }, "anthropic/claude-haiku-4-5": { "model_developer": "Anthropic", "model_type": "generalist", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", "cost_info": { "cost_per_1M_input_tokens": 1.0, "cost_per_1M_output_tokens": 5.0, "source": "Claude API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "anthropic/claude-sonnet-4-5": { "model_developer": "Anthropic", "model_type": "generalist", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", "cost_info": { "cost_per_1M_input_tokens": 3.0, "cost_per_1M_output_tokens": 15.0, "source": "Claude API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/claude-latency-rerun": { "model_developer": "Anthropic", "model_type": "generalist", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", "cost_info": { "cost_per_1M_input_tokens": 3.0, "cost_per_1M_output_tokens": 15.0, "source": "Claude API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "anthropic/claude-sonnet-4-6": { "model_developer": "Anthropic", "model_type": "generalist", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", "cost_info": { "cost_per_1M_input_tokens": 3.0, "cost_per_1M_output_tokens": 15.0, "source": "Claude API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "azure/analyze-text": { "model_developer": "Azure", "model_type": "specialized", "url": "https://learn.microsoft.com/en-us/azure/ai-services/content-safety/overview", "cost_info": { "cost_per_1M_input_tokens": 380.0, "cost_per_1M_output_tokens": 0.0, "source": "Azure AI Content Safety", "cost_per_h": "N/A", "additional_info": "Usage is measured in text records. One text record may contain up to 1000 characters. Only the input string is considered in the text record.\n1000 text records cost $0.38. Assuming an average length of 500 characters for a message, and a token length of approx. 4 characters: Estimated cost per 1M input tokens = 1M/(500/4) text records * $0.38/1000 text records) = $3.04.\nCost per 1M input tokens (not records) = 3.04$.\n\nThe benchmark was run on the free plan." }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "azure/promptshield": { "model_developer": "Azure", "model_type": "specialized", "url": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/content-filter-prompt-shields?view=foundry-classic#prompt-shields-for-user-prompts", "cost_info": { "cost_per_1M_input_tokens": 380.0, "cost_per_1M_output_tokens": 0.0, "source": "Azure AI Content Safety", "cost_per_h": "N/A", "additional_info": "Usage is measured in text records. One text record may contain up to 1000 characters. Only the input string is considered in the text record.\n1000 text records cost $0.38. Assuming an average length of 500 characters for a message, and a token length of approx. 4 characters: Estimated cost per 1M input tokens = 1M/(500/4) text records * $0.38/1000 text records) = $3.04.\nCost per 1M input tokens (not records) = 3.04$.\n\nThe benchmark was run on the free plan." }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "google/gemini-2.5-flash": { "model_developer": "Google", "model_type": "generalist", "url": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash", "cost_info": { "cost_per_1M_input_tokens": 0.3, "cost_per_1M_output_tokens": 2.5, "source": "Gemini Developer API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "lakera/lakera-guard_default": { "model_developer": "Lakera", "model_type": "specialized", "url": "https://www.lakera.ai/lakera-guard", "cost_info": { "cost_per_1M_input_tokens": 0.0, "cost_per_1M_output_tokens": 0.0, "source": "Lakera API pricing", "cost_per_h": "N/A", "additional_info": "Free for 10,000 API requests/month; Allows prompt size up to 8,000 tokens per request." }, "execution_specifications": { "type": "API", "details": "This is the LakeraGuard API with the default policy. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "meta/llama-guard-4-12b": { "model_developer": "Meta", "model_type": "specialized", "url": "https://www.llama.com/docs/model-cards-and-prompt-formats/llama-guard-4/", "cost_info": { "cost_per_1M_input_tokens": 0.2, "cost_per_1M_output_tokens": 0.0, "source": "Together AI pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "mistral/ministral-3b-2512": { "model_developer": "Mistral AI", "model_type": "generalist", "url": "https://docs.mistral.ai/models/ministral-3-3b-25-12", "cost_info": { "cost_per_1M_input_tokens": 0.1, "cost_per_1M_output_tokens": 0.1, "source": "Ministral API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "mistral/ministral-14b-2512": { "model_developer": "Mistral AI", "model_type": "generalist", "url": "https://docs.mistral.ai/models/ministral-3-14b-25-12", "cost_info": { "cost_per_1M_input_tokens": 0.2, "cost_per_1M_output_tokens": 0.2, "source": "Ministral API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "mistral/mistral-large-3": { "model_developer": "Mistral AI", "model_type": "generalist", "url": "https://docs.mistral.ai/models/mistral-large-3-25-12", "cost_info": { "cost_per_1M_input_tokens": 0.5, "cost_per_1M_output_tokens": 1.5, "source": "Mistral API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5-nano": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5-nano", "cost_info": { "cost_per_1M_input_tokens": 0.05, "cost_per_1M_output_tokens": 0.4, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to minimal; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5-mini": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5-mini", "cost_info": { "cost_per_1M_input_tokens": 0.25, "cost_per_1M_output_tokens": 2.0, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to low; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5.2": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5.2", "cost_info": { "cost_per_1M_input_tokens": 1.75, "cost_per_1M_output_tokens": 14.0, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to low; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5.2-informed": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5.2", "cost_info": { "cost_per_1M_input_tokens": 1.75, "cost_per_1M_output_tokens": 14.0, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to low; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5.2-uninformed": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5.2", "cost_info": { "cost_per_1M_input_tokens": 1.75, "cost_per_1M_output_tokens": 14.0, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to low; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-5.4": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://platform.openai.com/docs/models/gpt-5.4", "cost_info": { "cost_per_1M_input_tokens": 2.5, "cost_per_1M_output_tokens": 22.5, "source": "OpenAI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "Reasoning effort set to low; see OpenAI API docs. REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-oss-20b": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://www.together.ai/models/gpt-oss-20b", "cost_info": { "cost_per_1M_input_tokens": 0.05, "cost_per_1M_output_tokens": 0.2, "source": "Together AI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-oss-120b": { "model_developer": "OpenAI", "model_type": "generalist", "url": "https://www.together.ai/models/gpt-oss-120b", "cost_info": { "cost_per_1M_input_tokens": 0.05, "cost_per_1M_output_tokens": 0.2, "source": "Together AI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/gpt-oss-safeguard-20b": { "model_developer": "OpenAI", "model_type": "specialized", "url": "https://huggingface.co/openai/gpt-oss-safeguard-20b", "cost_info": { "cost_per_1M_input_tokens": 0.07, "cost_per_1M_output_tokens": 0.3, "source": "OpenRouter API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "openai/omni-moderation": { "model_developer": "OpenAI", "model_type": "specialized", "url": "https://platform.openai.com/docs/models/omni-moderation-latest", "cost_info": { "cost_per_1M_input_tokens": 0.0, "cost_per_1M_output_tokens": 0.0, "source": "OpenAI Moderation API pricing", "cost_per_h": "N/A", "additional_info": "The Omni Moderation endpoint is free to use but has low rate limits in the free tier. Your usage tier on the OpenAI API determines these rate limits." }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "virtue-ai/virtueguard-text-lite": { "model_developer": "Virtue AI", "model_type": "specialized", "url": "https://www.together.ai/models/virtueguard-text-lite", "cost_info": { "cost_per_1M_input_tokens": 0.2, "cost_per_1M_output_tokens": 0.0, "source": "Together AI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "x-ai/grok-4-1-fast-non-reasoning": { "model_developer": "X-AI", "model_type": "generalist", "url": "https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning", "cost_info": { "cost_per_1M_input_tokens": 0.2, "cost_per_1M_output_tokens": 0.5, "source": "X-AI API pricing", "cost_per_h": "N/A", "additional_info": "" }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "aws/bedrock-guardrail": { "model_developer": "Amazon Web Services", "model_type": "specialized", "url": "https://aws.amazon.com/bedrock/guardrails/", "cost_info": { "cost_per_1M_input_tokens": 150.0, "cost_per_1M_output_tokens": 0.0, "source": "Amazon Bedrock API pricing", "cost_per_h": "N/A", "additional_info": "Usage is measured in text units. One text unit may contain up to 1000 characters. Only the input string is considered in the text unit.\n1000 text units cost $0.15. Assuming an average length of 500 characters for a message, and a token length of approx. 4 characters: Estimated cost per 1M input tokens = 1M/(500/4) text units * $0.15/1000 text units) = $1.2.\nCost per 1M input tokens (not units) = 1.2$." }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "nvidia/aegis-ai-content-safety-llamaguard-defensive-1.0": { "model_developer": "NVIDIA", "model_type": "specialized", "url": "https://huggingface.co/nvidia/Aegis-AI-Content-Safety-LlamaGuard-Defensive-1.0", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "google/shieldgemma-2b": { "model_developer": "Google", "model_type": "specialized", "url": "https://huggingface.co/google/shieldgemma-2b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "google/shieldgemma-9b": { "model_developer": "Google", "url": "https://huggingface.co/google/shieldgemma-9b", "model_type": "specialized", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "google/shieldgemma-27b": { "model_developer": "Google", "model_type": "specialized", "url": "https://huggingface.co/google/shieldgemma-27b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "saillab/xguard": { "model_developer": "SAIL Lab", "model_type": "specialized", "url": "https://huggingface.co/saillab/x-guard", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "openai/gpt-oss-safeguard-120b": { "model_developer": "OpenAI", "model_type": "specialized", "url": "https://huggingface.co/openai/gpt-oss-safeguard-120b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 3.07, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (94GB NVL) on RunPod using vLLM." } }, "qwen3guard-gen-8b": { "model_developer": "Qwen", "model_type": "specialized", "url": "https://huggingface.co/Qwen/Qwen3Guard-Gen-8B", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "qwen/qwen3guard-gen-4b": { "model_developer": "Qwen", "model_type": "specialized", "url": "https://huggingface.co/Qwen/Qwen3Guard-Gen-4B", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "qwen/qwen3guard-gen-0.6b": { "model_developer": "Qwen", "model_type": "specialized", "url": "https://huggingface.co/Qwen/Qwen3Guard-Gen-0.6B", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "nvidia/llama-3.1-nemotron-safety-guard-8b-v3": { "model_developer": "NVIDIA", "model_type": "specialized", "url": "https://huggingface.co/nvidia/Llama-3.1-Nemotron-Safety-Guard-8B-v3", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "rakancorle1/thinkguard": { "model_developer": "RakanCorle1", "model_type": "specialized", "url": "https://huggingface.co/Rakancorle1/ThinkGuard", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "allenai/wildguard": { "model_developer": "AllenAI", "model_type": "specialized", "url": "https://huggingface.co/allenai/wildguard", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "toxicityprompts/polyguard-ministral": { "model_developer": "ToxicityPrompts", "model_type": "specialized", "url": "https://huggingface.co/ToxicityPrompts/PolyGuard-Ministral", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "toxicityprompts/polyguard-qwen": { "model_developer": "ToxicityPrompts", "model_type": "specialized", "url": "https://huggingface.co/ToxicityPrompts/PolyGuard-Qwen", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "toxicityprompts/polyguard-qwen-smol": { "model_developer": "ToxicityPrompts", "model_type": "specialized", "url": "https://huggingface.co/ToxicityPrompts/PolyGuard-Qwen-Smol", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.0-2b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.0-2b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.0-8b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.0-8b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.1-2b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.1-2b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.1-8b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.1-8b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.2-5b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.2-5b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.2-3b-a800m": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.2-3b-a800m", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "ibm-granite/granite-guardian-3.3-8b": { "model_developer": "IBM", "model_type": "specialized", "url": "https://huggingface.co/ibm-granite/granite-guardian-3.3-8b", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using vLLM." } }, "govtech/lionguard-2": { "model_developer": "GovTech", "model_type": "specialized", "url": "https://huggingface.co/govtech/lionguard-2", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "govtech/lionguard-2.1": { "model_developer": "GovTech", "model_type": "specialized", "url": "https://huggingface.co/govtech/lionguard-2.1", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "govtech/lionguard-2-lite": { "model_developer": "GovTech", "model_type": "specialized", "url": "https://huggingface.co/govtech/lionguard-2-lite", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "leolee99/piguard": { "model_developer": "Leolee99", "model_type": "specialized", "url": "https://huggingface.co/leolee99/PIGuard", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "meta-llama/llama-prompt-guard-2-22m": { "model_developer": "Meta", "model_type": "specialized", "url": "https://huggingface.co/meta-llama/Llama-Prompt-Guard-2-22M", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "meta-llama/llama-prompt-guard-2-86m": { "model_developer": "Meta", "model_type": "specialized", "url": "https://huggingface.co/meta-llama/Llama-Prompt-Guard-2-86M", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the transformers Library." } }, "protectai/llm-guard": { "model_developer": "ProtectAI", "model_type": "specialized", "url": "https://protectai.github.io/llm-guard/input_scanners/prompt_injection/", "cost_info": { "cost_per_1M_input_tokens": "N/A", "cost_per_1M_output_tokens": "N/A", "source": "RunPod", "cost_per_h": 2.39, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an H100 (80GB PCIe) on RunPod using the llm-guard Python SDK." } }, "neuraltrust/promptguard": { "model_developer": "NeuralTrust", "model_type": "specialized", "url": "https://neuraltrust.ai/prompt-guard", "cost_info": { "cost_per_1M_input_tokens": 0.0, "cost_per_1M_output_tokens": 0.0, "source": "NeuralTrust", "cost_per_h": "N/A", "additional_info": "Usage is free on the API. We were banned after about 1800 requests." }, "execution_specifications": { "type": "API", "details": "REST API accessed through a CPU3 RunPod instance at US-KS-2." } }, "bells-o/opencc-jb-escalation": { "model_developer": "CeSIA", "model_type": "specialized", "url": "https://huggingface.co/centrepourlasecuriteia/opencc-jb-escalation", "cost_info": { "cost_per_1M_input_tokens": 0.0, "cost_per_1M_output_tokens": 0.0, "source": "RunPod", "cost_per_h": 1.49, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an A100 (80GB PCIe) on RunPod using vLLM." } }, "bells-o/opencc-cm-escalation": { "model_developer": "CeSIA", "model_type": "specialized", "url": "https://huggingface.co/centrepourlasecuriteia/opencc-cm-escalation", "cost_info": { "cost_per_1M_input_tokens": 0.0, "cost_per_1M_output_tokens": 0.0, "source": "RunPod", "cost_per_h": 1.49, "additional_info": "Output token cost is disregarded. The cost per 1M input tokens is estimated as total_cost * (1,000,000 / total_input_tokens)." }, "execution_specifications": { "type": "Local", "details": "This model was ran on an A100 (80GB PCIe) on RunPod using vLLM." } } }