{
  "schema_version": "1",
  "name": "Clipia",
  "description": "Generate AI images, video, speech, music and presentations; chat with text models, plan scenes, compose finished videos, browse models and prompt templates, and check your Clipia credit balance.",
  "mcp_server": "https://mcp.clipia.ai/mcp",
  "documentation": "https://clipia.ai/en/docs/mcp",
  "authentication": "https://clipia.ai/auth.md",
  "skills": [
    {
      "id": "generate_image",
      "name": "Generate image",
      "description": "Generate image(s) from a text prompt, optionally with reference images (editing / image-to-image). Waits briefly and usually returns the finished image inline. Cost in credits is returned.",
      "inputs": {
        "prompt": "string (required)",
        "model": "string (optional)",
        "aspect_ratio": "string e.g. '1:1', '16:9', '9:16'",
        "resolution": "string e.g. '1K', '2K', '4K'",
        "num_images": "integer 1-4",
        "image_urls": "array[string] max 4 — reference images"
      }
    },
    {
      "id": "generate_video",
      "name": "Generate video",
      "description": "Start a video generation from a text prompt (text-to-video) or a start image (image-to-video). Returns request_id and cost immediately — renders take 1–10 min; poll with wait_generation.",
      "inputs": {
        "prompt": "string (required, English recommended)",
        "model": "string (optional)",
        "duration": "number — clip length in seconds",
        "aspect_ratio": "string e.g. '16:9', '9:16'",
        "resolution": "string e.g. '480p', '720p', '1080p'",
        "image_url": "string — start frame for image-to-video"
      }
    },
    {
      "id": "generate_audio",
      "name": "Generate audio",
      "description": "Generate speech from text (text-to-speech): pick a voice and language, get an mp3 back. Returns request_id and cost; poll with wait_generation until COMPLETED for output.audio.url. Use the request_id as a voiceover in compose_video, or the mp3 on its own.",
      "inputs": {
        "text": "string (required) — Russian or English, typically ≤500 chars",
        "model": "string (optional) — audio model slug",
        "voice": "string (optional) — voice preset (see get_model)",
        "language": "string (optional) e.g. 'ru', 'en'"
      }
    },
    {
      "id": "generate_music",
      "name": "Generate music",
      "description": "Generate background music / a soundtrack from a text description (mood, genre, tempo, instruments). Instrumental by default — ideal under narration. Returns request_id and cost; poll with wait_generation until COMPLETED for output.audio.url. Use the request_id as audio_request_id in compose_video, or the mp3 on its own.",
      "inputs": {
        "prompt": "string (required, English recommended) — mood/genre/tempo/instruments",
        "instrumental": "boolean (optional, default true) — false allows vocals",
        "model": "string (optional) — music model slug"
      }
    },
    {
      "id": "chat",
      "name": "Chat with an LLM",
      "description": "Chat with a text LLM and get the reply text plus token usage and credit cost. Pass either a single prompt or a full messages array. Charged in credits from the connected account.",
      "inputs": {
        "model": "string (optional) — LLM model slug",
        "prompt": "string — single user message",
        "messages": "array — {role, content} conversation messages",
        "max_tokens": "integer (optional) — reply length cap",
        "temperature": "number (optional) — sampling temperature"
      }
    },
    {
      "id": "generate_scenario",
      "name": "Generate scenario",
      "description": "Plan a multi-scene video from a brief: an LLM director returns per-scene English prompts with durations, plus a soundtrack prompt. Feed each scene prompt to generate_video, then stitch with compose_video. Charged in credits like chat.",
      "inputs": {
        "prompt": "string (required) — the brief / idea for the video",
        "duration_seconds": "number (optional) — target total length",
        "scene_count": "integer (optional) — number of scenes"
      }
    },
    {
      "id": "compose_video",
      "name": "Compose video",
      "description": "Stitch 2–20 finished video scenes into ONE final clip server-side (normalized frame/fps, concatenated, optional voiceover and soundtrack, optional burned-in subtitles). Requires idempotency_key. Reuse the same key only when retrying the identical unchanged request after a transport or ambiguous failure; use a new key for a changed request. Returns request_id (cmp_*) — poll with wait_generation; output.video.url is the final mp4.",
      "inputs": {
        "scenes": "array 2-20 — {request_id} of a COMPLETED video (or {video_url}) in playback order",
        "audio_request_id": "string (optional) — soundtrack from a COMPLETED audio generation",
        "voiceover_request_id": "string (optional) — narration mixed over the video",
        "subtitles": "array (optional) — {text, start_ms, end_ms} burned into the final clip",
        "aspect_ratio": "string e.g. '9:16', '16:9', '1:1'",
        "idempotency_key": "string (required, 8–128 chars) — stable key for this exact request; reuse only for an identical retry"
      }
    },
    {
      "id": "wait_generation",
      "name": "Wait for generation",
      "description": "Long-poll a generation until COMPLETED, FAILED or CANCELED. Returns output URLs (and an inline preview) when done.",
      "inputs": {
        "request_id": "string (required)",
        "wait_seconds": "integer 1-30 (default 25)"
      }
    },
    {
      "id": "get_generation",
      "name": "Get generation",
      "description": "Get the current status/result of a generation without waiting. When COMPLETED returns output.images[].url (preview) and original_url (full quality).",
      "inputs": {
        "request_id": "string (required)"
      }
    },
    {
      "id": "list_models",
      "name": "List models",
      "description": "List available AI models with type (text/image/video/audio), capabilities and pricing in credits.",
      "inputs": {
        "type": "string enum 'text'|'image'|'video'|'audio'",
        "search": "string — substring filter",
        "limit": "integer 1-100 (default 50)"
      }
    },
    {
      "id": "get_model",
      "name": "Get model",
      "description": "Get model details and pricing: input schema for generation, context and per-token rates for text models.",
      "inputs": {
        "model": "string (required) — model slug"
      }
    },
    {
      "id": "get_balance",
      "name": "Get balance",
      "description": "Get the credit balance of the connected account and 30-day usage of the API key.",
      "inputs": {}
    },
    {
      "id": "search_templates",
      "name": "Search templates",
      "description": "Search 3500+ curated prompt templates (hybrid text+semantic, RU or EN). Each result has a ready prompt and a recommended model.",
      "inputs": {
        "query": "string (required)",
        "media_type": "string enum 'image'|'video'",
        "limit": "integer 1-20 (default 8)"
      }
    },
    {
      "id": "generate_presentation",
      "name": "Generate presentation",
      "description": "Render a NEW slide deck (editable PPTX + PDF + PNG previews) from a DeckSpec you compose: title, language (ru/en), theme picked to fit the subject, optional brand, and 3–20 slides. Each slide has a layout (cover, section, bullets, two-col, image-full, quote, stats, closing); set image.prompt (English) for an AI illustration or skip for text-only. Returns request_id (prs_*) — poll with wait_generation; output has pptx_url, pdf_url, preview_urls[]. Costs credits (render fee + illustrations).",
      "inputs": {
        "title": "string (required) — cover title",
        "language": "string (optional) — 'ru' or 'en'",
        "theme": "string (optional) — dark: 'midnight' (default), 'slate', 'noir', 'forest', 'ember'; light: 'paper', 'ivory', 'azure'. Pick by subject; 'clipia-dark'/'clean-light' are Clipia brand palettes",
        "brand": "object (optional) — {name?, website?} of the presenter; omit for unbranded slides",
        "slides": "array 3-20 (required) — each {layout, title?, bullets?[], image?:{prompt}, notes?}"
      }
    },
    {
      "id": "edit_presentation",
      "name": "Edit presentation",
      "description": "Edit an existing deck instead of regenerating it: change slide text, title, theme, add/remove/reorder slides. The deck is rebuilt from its saved spec and previously generated illustrations are reused without another illustration charge — a text fix costs only the render fee. Indexes are 0-based positions in the current deck. Illustrations are redrawn (and billed) only for slides whose image.prompt changed or that you list in regenerate_images. Returns request_id (prs_*) — poll with wait_generation.",
      "inputs": {
        "work_id": "string (optional) — deck to edit; omit for the latest one",
        "title": "string (optional) — new deck title",
        "language": "string (optional) — 'ru' or 'en'",
        "theme": "string (optional) — new theme slug",
        "slides": "array (optional) — operations {op: set|insert|delete|move, index, to_index?, slide?}",
        "regenerate_images": "integer[] (optional) — slide indexes to redraw",
        "regenerate_all_images": "boolean (optional) — redraw every illustration"
      }
    }
  ]
}