{"id":4211,"external_id":"plugins_6a57ef89f2d481918513e6133ec3fc18","name":"amd-skills","display_name":"AMD","developer":"AMD","category":"Developer Tools","listing_language":"en","listing_language_details":{"method":"cld-0.13.0/listing-v1","reliable":true,"detected_at":"2026-10-01T13:22:03Z","input_sha256":"f45ffdfde1fc839063a6890fed82785e8753cce0e8a2f179ad60b053ebc215b2","source_fields":["release.description","release.interface.short_description","release.interface.long_description"]},"version":"0.2.0","skill_count":5,"first_seen_at":"2026-09-30T22:02:35.000Z","last_seen_at":"2026-10-01T18:00:02.555Z","last_changed_at":"2026-09-30T22:02:35.000Z","install_count":null,"research_summary":null,"research_reviewed_at":null,"metadata":{"id":"plugins_6a57ef89f2d481918513e6133ec3fc18","name":"amd-skills","scope":"GLOBAL","status":"ENABLED","release":{"id":"pluginrel_fca7d37185088191ba5d21cf97aa45af","skills":[{"name":"hyperloom-workload-optimizer","interface":{"brand_color":null,"iconography":"radar","display_name":"hyperloom-workload-optimizer","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Autonomously optimizes end-to-end LLM inference throughput on AMD Instinct GPUs and reports a validated gain, using the Hyperloom multi-agent optimizer. Given a model, framework, workload (TP/EP, concurrency, ISL/OSL, precision), an objective and a time budget, it explores per-workload which levers to pull (serving/config parameters and env, framework enablement and source patches, and hot GPU-kernel rewrites), benchmarks each candidate, and returns the optimization stack that produced the gain. Use when the user wants to make a model serve faster, raise tokens/sec or throughput, optimize or tune vLLM or SGLang on MI300X/MI325X/MI355X, run Hyperloom, run the kernel-agent, quantize-then-optimize with Quark, set up Hyperloom from scratch, or resume a Hyperloom session. Do not use to stand up a server for plain serving, diagnose a broken ROCm install, or run a one-off kernel/benchmark or trace analysis without the optimization loop."},"description":"Autonomously optimizes end-to-end LLM inference throughput on AMD Instinct GPUs and reports a validated gain, using the Hyperloom multi-agent optimizer. Given a model, framework, workload (TP/EP, concurrency, ISL/OSL, precision), an objective and a time budget, it explores per-workload which levers to pull (serving/config parameters and env, framework enablement and source patches, and hot GPU-kernel rewrites), benchmarks each candidate, and returns the optimization stack that produced the gain. Use when the user wants to make a model serve faster, raise tokens/sec or throughput, optimize or tune vLLM or SGLang on MI300X/MI325X/MI355X, run Hyperloom, run the kernel-agent, quantize-then-optimize with Quark, set up Hyperloom from scratch, or resume a Hyperloom session. Do not use to stand up a server for plain serving, diagnose a broken ROCm install, or run a one-off kernel/benchmark or trace analysis without the optimization loop.","plugin_release_skill_id":"pluginrsk_6a8de0ffabcc8191a664c75e99432df9"},{"name":"local-ai-app-integration","interface":{"brand_color":null,"iconography":"code","display_name":"local-ai-app-integration","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Integrates local AI capabilities into applications using Embeddable Lemonade. Use when the user wants to add local AI, offline AI, private AI, on-device AI, a local LLM, local chat, embeddings, image generation, speech-to-text, or text-to-speech to an existing app; replace or supplement OpenAI, Anthropic, Ollama, or other cloud AI APIs with a local backend; only use to convert user apps. Do not use when the user just wants the agent itself to generate images, transcribe, or speak locally in the current workspace, even to cut their own API bill."},"description":"Integrates local AI capabilities into applications using Embeddable Lemonade. Use when the user wants to add local AI, offline AI, private AI, on-device AI, a local LLM, local chat, embeddings, image generation, speech-to-text, or text-to-speech to an existing app; replace or supplement OpenAI, Anthropic, Ollama, or other cloud AI APIs with a local backend; only use to convert user apps. Do not use when the user just wants the agent itself to generate images, transcribe, or speak locally in the current workspace, even to cut their own API bill.","plugin_release_skill_id":"pluginrsk_6a8de0ffab2c819191dafc4f5468ce4b"},{"name":"local-ai-use","interface":{"brand_color":null,"iconography":"hierarchy","display_name":"local-ai-use","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Makes this agent generate images, transcribe audio, and synthesize speech on the user's own machine through a local Lemonade Server instead of a paid cloud API. Use it above all to change that routing persistently, from now on — keep generating pictures locally while chat stays on the cloud; set this workspace up to make images on my own machine — even when the user asks for no image or file in the same breath. Also use it for a single request the user wants done locally, offline, on-device, or kept private: transcribe this recording, make this picture, read this text aloud. Applies in Claude, Cursor, Codex, or any agent harness. Use when the user wants to cut cost or tokens on image, audio, or voice API calls, or to drop DALL-E, hosted Whisper, ElevenLabs, or other paid multimodal APIs; or mentions Lemonade Server, OmniRouter, SD-Turbo, kokoro, Ryzen AI, or NPU/iGPU/dGPU inference. Changes no application source code; do not use it if the user is adding local AI to an app they ship."},"description":"Makes this agent generate images, transcribe audio, and synthesize speech on the user's own machine through a local Lemonade Server instead of a paid cloud API. Use it above all to change that routing persistently, from now on — keep generating pictures locally while chat stays on the cloud; set this workspace up to make images on my own machine — even when the user asks for no image or file in the same breath. Also use it for a single request the user wants done locally, offline, on-device, or kept private: transcribe this recording, make this picture, read this text aloud. Applies in Claude, Cursor, Codex, or any agent harness. Use when the user wants to cut cost or tokens on image, audio, or voice API calls, or to drop DALL-E, hosted Whisper, ElevenLabs, or other paid multimodal APIs; or mentions Lemonade Server, OmniRouter, SD-Turbo, kokoro, Ryzen AI, or NPU/iGPU/dGPU inference. Changes no application source code; do not use it if the user is adding local AI to an app they ship.","plugin_release_skill_id":"pluginrsk_6a8de0ffac84819193cbff3112db93ff"},{"name":"serving-llms-on-instinct","interface":{"brand_color":null,"iconography":"code","display_name":"serving-llms-on-instinct","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Serves AI models on AMD Instinct GPU hardware using vLLM. Use this skill whenever the user wants to run, serve, deploy, start, host, or launch a language model on an AMD GPU, AMD Instinct, MI300X, MI325X, MI350X, or MI355X. Also use when the user mentions vLLM on ROCm, vLLM on AMD, serving on HBM, or asks how to get a model running on AMD data center hardware. Use when the user asks \"run Qwen3\", \"serve DeepSeek\", \"start a vLLM endpoint\", \"get a model running on my AMD machine\", or any similar phrasing. Handles the full flow: GPU detection, environment validation, vLLM configuration, launch, and health verification. Do not use for NVIDIA GPUs, consumer AMD GPUs (RX series, Radeon), Ryzen AI, NPU, MI250X, or MI100."},"description":"Serves AI models on AMD Instinct GPU hardware using vLLM. Use this skill whenever the user wants to run, serve, deploy, start, host, or launch a language model on an AMD GPU, AMD Instinct, MI300X, MI325X, MI350X, or MI355X. Also use when the user mentions vLLM on ROCm, vLLM on AMD, serving on HBM, or asks how to get a model running on AMD data center hardware. Use when the user asks \"run Qwen3\", \"serve DeepSeek\", \"start a vLLM endpoint\", \"get a model running on my AMD machine\", or any similar phrasing. Handles the full flow: GPU detection, environment validation, vLLM configuration, launch, and health verification. Do not use for NVIDIA GPUs, consumer AMD GPUs (RX series, Radeon), Ryzen AI, NPU, MI250X, or MI100.","plugin_release_skill_id":"pluginrsk_6a8de100a96c81919c6501ef98c77b96"},{"name":"tracelens-analysis-orchestrator","interface":{"brand_color":null,"iconography":"chart","display_name":"tracelens-analysis-orchestrator","default_prompt":null,"icon_large_url":null,"icon_small_url":null,"short_description":"Orchestrates modular PyTorch profiler trace analysis with TraceLens: generates perf reports, prepares category data, runs system-level and compute-kernel subagents in parallel, validates outputs, and writes a prioritized stakeholder report (analysis.md). Use when the user asks to follow the analysis orchestrator, run the agentic analysis workflow, analyze a trace, compare two traces, or mentions standalone or comparative TraceLens analysis."},"description":"Orchestrates modular PyTorch profiler trace analysis with TraceLens: generates perf reports, prepares category data, runs system-level and compute-kernel subagents in parallel, validates outputs, and writes a prioritized stakeholder report (analysis.md). Use when the user asks to follow the analysis orchestrator, run the agentic analysis workflow, analyze a trace, compare two traces, or mentions standalone or comparative TraceLens analysis.","plugin_release_skill_id":"pluginrsk_6a8de0ffa9a08191a6f0c450ed326c02"}],"app_ids":[],"version":"0.2.0","keywords":["amd","rocm","hip","ryzen-ai","vllm","lemonade","local-ai"],"interface":{"category":"Developer Tools","logo_url":"https://files.openai.com/content?id=file_00000000b23081f6b00e59ebc98c3c4c","brand_color":"#ED1C24","website_url":"https://github.com/amd/skills","capabilities":["Read","Write"],"logo_url_dark":null,"default_prompt":"Use AMD Skills to deploy this LLM for inference on my AMD Instinct GPU","developer_name":"AMD","default_prompts":["Use AMD Skills to deploy this LLM for inference on my AMD Instinct GPU","Learn how to generate images locally and generate the image of a cat","Convert my cloud LLM app into an app that uses local inference"],"screenshot_urls":[],"long_description":"AMD's verified Agent Skills in one plugin: route image/audio through local AI on Ryzen AI, serve LLMs on AMD Instinct GPUs with vLLM, optimize inference throughput with Hyperloom, and analyze GPU kernel and PyTorch trace performance.","composer_icon_url":"https://files.openai.com/content?id=file_0000000064c481f6ae76ceeba6e85863","short_description":"Enable AMD's skills ecosystem","plugin_category_id":"developer tools","privacy_policy_url":null,"terms_of_service_url":null,"composer_icon_dark_url":null},"description":"Agent Skills for AMD-optimized workflows.","app_manifest":null,"display_name":"AMD","app_templates":[],"onboarding_skill_name":null,"requires_local_executor":false},"created_at":"2026-07-16T18:03:32.264346Z","is_template":false,"connector_id":null,"discoverability":"UNLISTED","canonical_app_id":null},"research":null,"package_metadata":{"name":"amd-skills","author":{"name":"AMD"},"license":"MIT","sources":[{"path":".codex-plugin/plugin.json","sha256":"cbe68f54a64025845974c6d485a9b765a307f70548952b61be30b67e1835ac49"}],"version":"0.2.0","homepage":"https://github.com/amd/skills","keywords":["amd","rocm","hip","ryzen-ai","vllm","lemonade","local-ai"],"repository":"https://github.com/amd/skills","artifact_id":5102,"observed_at":"2026-09-30T23:13:27Z","capabilities":["Read","Write"],"field_sources":{"name":0,"author":0,"license":0,"version":0,"homepage":0,"keywords":0,"repository":0,"capabilities":0},"extraction_version":1,"conflicts_or_errors":[]}}