{"id":4496,"external_id":"plugins_6a8efd9d54688191b59f7107a44c8194","name":"cerebrium","display_name":"Cerebrium","developer":"Cerebrium","category":"Other","listing_language":"en","listing_language_details":{"method":"cld-0.13.0/listing-v1","reliable":true,"detected_at":"2026-10-01T13:22:04Z","input_sha256":"43c7539193a14ae12e566a6bc38c044772558ef96764f6f8768e0ac4c1b11a59","source_fields":["release.description","release.interface.short_description","release.interface.long_description"]},"version":"0.1.0","skill_count":1,"first_seen_at":"2026-09-30T22:02:35.000Z","last_seen_at":"2026-10-01T12:00:01.032Z","last_changed_at":"2026-09-30T22:02:35.000Z","install_count":null,"research_summary":null,"research_reviewed_at":null,"metadata":{"id":"plugins_6a8efd9d54688191b59f7107a44c8194","name":"cerebrium","scope":"GLOBAL","status":"ENABLED","release":{"id":"pluginrel_1ea52e948ae88191bd60dd5b6236bb00","skills":[{"name":"cerebrium","interface":{"brand_color":null,"iconography":"code","display_name":"cerebrium","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Use for any Cerebrium task: deploying Python code to serverless GPU or CPU, writing or fixing cerebrium.toml, choosing hardware and regions, calling deployed endpoints (REST, streaming, WebSocket, async), autoscaling and concurrency, cold starts, secrets, CI/CD, and debugging a build or a running app from the terminal. Covers the cerebrium CLI, configuration defaults the API actually applies, accepted GPU identifiers with per-plan limits, and troubleshooting."},"description":"Use for any Cerebrium task: deploying Python code to serverless GPU or CPU, writing or fixing cerebrium.toml, choosing hardware and regions, calling deployed endpoints (REST, streaming, WebSocket, async), autoscaling and concurrency, cold starts, secrets, CI/CD, and debugging a build or a running app from the terminal. Covers the cerebrium CLI, configuration defaults the API actually applies, accepted GPU identifiers with per-plan limits, and troubleshooting.","plugin_release_skill_id":"pluginrsk_6a8efda0029c81919c29a9141519c97e"}],"app_ids":[],"version":"0.1.0","keywords":["serverless","gpu","inference","deployment","machine-learning","llm","vllm","autoscaling","cold-start","streaming","websockets","cerebrium"],"interface":{"category":"Other","logo_url":"https://files.openai.com/content?id=file_0000000018c08211b27bc672d81a4b1c","brand_color":null,"website_url":"https://www.cerebrium.ai","capabilities":["Read, Write"],"logo_url_dark":null,"default_prompt":"Deploy this Python inference function to a serverless GPU and give me the endpoint to call.","developer_name":"Cerebrium","default_prompts":["Deploy this Python inference function to a serverless GPU and give me the endpoint to call.","My app is queueing requests under load. Diagnose the scaling config and fix it.","Convert this FastAPI app to run on Cerebrium with a WebSocket endpoint."],"screenshot_urls":[],"long_description":"Cerebrium runs Python workloads on serverless GPU and CPU with scale to zero and per second billing: REST endpoints, SSE streaming, WebSockets and async jobs, all described by one cerebrium.toml and driven by one CLI.\n\nThis plugin helps you go from a Python function to a deployed endpoint, and then keep it healthy. It covers choosing hardware, regions and the right runtime, writing and fixing cerebrium.toml, picking a scaling metric that matches the workload, calling the endpoint over REST, streaming, WebSocket or async, handling secrets and environment variables, wiring CI/CD with a service account, and debugging a failed build or an app that is queueing or returning 5xx.\n\nIt is built for inference APIs for language, embedding and vision models, real time voice and video apps, bursty traffic that should not hold idle GPUs, multi region deployments, and migrations from Replicate, Hugging Face or Mystic.","composer_icon_url":"https://files.openai.com/content?id=file_000000001c7c8208af4e93cd953a493c","short_description":"Deploy AI on serverless GPUs","plugin_category_id":"other","privacy_policy_url":"https://www.cerebrium.ai/privacy","terms_of_service_url":"https://www.cerebrium.ai/terms-of-service","composer_icon_dark_url":null},"description":"Official Cerebrium agent skills and hosted docs MCP: deploy and operate real-time AI workloads on serverless GPU and CPU (REST, streaming, WebSocket and async endpoints), with verified cerebrium.toml, hardware and autoscaling guidance.","app_manifest":null,"display_name":"Cerebrium","app_templates":[],"onboarding_skill_name":null,"requires_local_executor":false},"created_at":"2026-08-26T15:05:25.768646Z","is_template":false,"connector_id":null,"discoverability":"UNLISTED","canonical_app_id":null},"research":null,"package_metadata":{"name":"cerebrium","author":{"url":"https://github.com/CerebriumAI","name":"Cerebrium"},"license":"MIT","sources":[{"path":".codex-plugin/plugin.json","sha256":"51476c2aa9a3407bee4a7c20ae98a502fc0cfc8708178ecd0fe55d00238780c5"},{"path":".claude-plugin/plugin.json","sha256":"ab364ab9cf66f6cc2214d9882f9a593eb53595d1c0925f169626073eee944eb5"}],"version":"0.1.0","homepage":"https://www.cerebrium.ai","keywords":["serverless","gpu","inference","deployment","machine-learning","llm","vllm","autoscaling","cold-start","streaming","websockets","cerebrium"],"repository":"https://github.com/CerebriumAI/cerebrium-skills","artifact_id":5398,"observed_at":"2026-09-30T23:15:03Z","support_url":"https://www.cerebrium.ai/contact","capabilities":["Read, Write"],"field_sources":{"name":0,"author":0,"license":0,"version":0,"homepage":0,"keywords":0,"repository":0,"support_url":0,"capabilities":0},"extraction_version":1,"conflicts_or_errors":[]}}