{
  "skill_name": "ad-creative",
  "evals": [
    {
      "id": 1,
      "prompt": "Generate ad creative for our Meta (Facebook/Instagram) campaign. We sell an AI writing assistant for content marketers. Main value prop: write blog posts 5x faster. Target audience: content marketing managers at B2B SaaS companies. Budget: $5k/month.",
      "expected_output": "Should check for product-marketing.md first. Should generate creative following the angle-based approach: identify 3-5 angles (speed, quality, ROI, pain of blank page, competitive edge). For each angle, should generate primary text (≤125 chars), headline (≤40 chars), and description (≤30 chars) respecting Meta character limits. Should provide multiple variations per angle. Should suggest image/visual direction for each. Should organize output with angle name, hook, body, CTA for each variation. Should recommend which angles to test first.",
      "assertions": [
        "Checks for product-marketing.md",
        "Uses angle-based generation approach",
        "Identifies multiple angles (3-5)",
        "Respects Meta character limits (125/40/30)",
        "Generates multiple variations per angle",
        "Suggests image or visual direction",
        "Includes hook, body, and CTA for each",
        "Recommends which angles to test first"
      ],
      "files": []
    },
    {
      "id": 2,
      "prompt": "I need Google Ads copy for our CRM product. We're targeting the keyword 'best CRM for small business'. Need responsive search ads.",
      "expected_output": "Should generate Google RSA creative respecting character limits: headlines (≤30 chars each, need 10-15 variations) and descriptions (≤90 chars each, need 4+ variations). Should note that pinning should be used sparingly as it reduces optimization. Should include the target keyword in headlines. Should provide multiple angle-based variations. Should suggest ad extensions (sitelinks, callouts, structured snippets). Should follow Google Ads best practices for RSA.",
      "assertions": [
        "Respects Google RSA character limits (30 char headlines, 90 char descriptions)",
        "Generates 10-15 headline variations",
        "Generates 4+ description variations",
        "Includes target keyword in headlines",
        "Notes pinning should be used sparingly per skill guidance",
        "Suggests ad extensions",
        "Uses angle-based variation approach"
      ],
      "files": []
    },
    {
      "id": 3,
      "prompt": "Here's our ad performance data: Ad A (pain point angle) - CTR 2.1%, CPC $3.20, Conv rate 4.5%. Ad B (social proof angle) - CTR 1.4%, CPC $4.10, Conv rate 6.2%. Ad C (feature angle) - CTR 0.8%, CPC $5.50, Conv rate 2.1%. Help me iterate on these.",
      "expected_output": "Should activate the iteration-from-performance mode (not generate-from-scratch). Should analyze the data: Ad A has best CTR, Ad B has best conversion rate (highest efficiency despite lower CTR), Ad C is underperforming on all metrics. Should recommend doubling down on the pain point angle (high CTR) and social proof angle (high conversion), while pausing or reworking the feature angle. Should generate new variations that combine winning elements (pain point hook + social proof). Should suggest specific iterations on Ad A and Ad B.",
      "assertions": [
        "Activates iteration mode based on performance data",
        "Analyzes CTR, CPC, and conversion rate for each ad",
        "Identifies winning angles from the data",
        "Recommends pausing or reworking underperforming creative",
        "Generates new variations combining winning elements",
        "Provides specific iterations on top performers"
      ],
      "files": []
    },
    {
      "id": 4,
      "prompt": "we need linkedin ads for our enterprise security product. audience is CISOs and IT directors.",
      "expected_output": "Should trigger on casual phrasing. Should generate LinkedIn ad creative respecting character limits: introductory text (≤150 chars), headline (≤70 chars), description (≤100 chars). Should adapt tone and messaging for enterprise security audience (CISOs, IT directors) — more formal, compliance-focused, risk-reduction language. Should provide multiple angles relevant to security buyers (risk reduction, compliance, incident response time, cost of breaches). Should suggest ad format recommendations for LinkedIn (sponsored content, message ads, etc.).",
      "assertions": [
        "Triggers on casual phrasing",
        "Respects LinkedIn character limits (150/70/100)",
        "Adapts tone for enterprise security audience",
        "Uses risk-reduction and compliance language",
        "Provides multiple angles relevant to security buyers",
        "Suggests LinkedIn ad format recommendations"
      ],
      "files": []
    },
    {
      "id": 5,
      "prompt": "I need to generate a big batch of ad variations for a multi-platform campaign launching next week. We're a meal delivery service targeting busy professionals. Need ads for Google, Meta, and TikTok.",
      "expected_output": "Should activate the batch generation workflow. Should generate creative for all three platforms respecting each platform's character limits: Google RSA (30/90), Meta (125/40/30), TikTok (80 chars recommended, 100 max). Should identify 3-5 angles that work across platforms (convenience, health, time savings, variety, cost vs eating out). Should generate variations per angle per platform. Should note platform-specific creative considerations (TikTok needs video concepts, not just text). Should organize output clearly by platform.",
      "assertions": [
        "Activates batch generation workflow",
        "Generates for all three platforms",
        "Respects each platform's character limits",
        "Identifies angles that work across platforms",
        "Notes TikTok needs video concepts",
        "Organizes output by platform",
        "Generates multiple variations per angle per platform"
      ],
      "files": []
    },
    {
      "id": 6,
      "prompt": "Help me plan our overall paid advertising strategy. We have a $20k monthly budget and want to figure out which platforms to use and how to allocate spend.",
      "expected_output": "Should recognize this is a paid advertising strategy task, not ad creative generation. Should defer to or cross-reference the ads skill, which handles campaign strategy, platform selection, and budget allocation. May briefly mention creative considerations but should make clear that ads is the right skill for strategy.",
      "assertions": [
        "Recognizes this as paid ads strategy, not creative generation",
        "References or defers to ads skill",
        "Does not attempt full campaign strategy using creative generation patterns"
      ],
      "files": []
    },
    {
      "id": 7,
      "prompt": "I want to make one of those iMessage-style video ads for Meta — the ones where a fake text conversation reveals the product and a promo code. We sell a sleep tracking ring. Our promo code is RESTED.",
      "expected_output": "Should load references/imessage-video-ads.md. Should start by picking a concept angle from the six-angle catalog (result-as-screenshot, setup flex, cancellation moment, feature-as-punchline, friend-asks-friend inverse, receipt-as-hook) before writing bubbles — likely result-as-screenshot (a sleep score) for this product. Should draft an 8-14 bubble script in real texting voice where the brand appears only after the peer asks, with the RESTED code delivered conversationally inside a bubble and repeated on a static end card. Should apply grounding rules: any sleep-improvement claim in the thread must trace to a real customer result or product fact, and the thread must not be framed as a real testimonial. Should present production route options (off-the-shelf skill, Playwright+ffmpeg pipeline, or Remotion) rather than assuming one, and mention key craft rules (the recognizable send/receive SFX, silent typing indicators, 9:16 1080x1920).",
      "assertions": [
        "Loads or applies the imessage-video-ads reference",
        "Selects a concept angle before writing the script",
        "Script is 8-14 bubbles in authentic texting voice",
        "Brand name appears only after the peer asks about it",
        "Promo code RESTED appears in a bubble and on the end card",
        "Applies grounding rules — no fabricated claims, not framed as a real testimonial",
        "Mentions at least one production route and key craft rules (SFX, silent typing indicator, 9:16)"
      ],
      "files": []
    },
    {
      "id": 8,
      "prompt": "We sell a menopause supplement. I saw those ads where someone asks ChatGPT a health question and the answer recommends the product — make one of those for us. Also curious about the Apple Notes version.",
      "expected_output": "Should load references/imessage-video-ads.md and apply the Other iOS-Native Reveal Surfaces section. Should flag the compliance constraint prominently BEFORE drafting: a fabricated AI answer making health claims is the highest-risk version of this format — every claim needs substantiation, health/medical advice in a fake ChatGPT answer needs legal review, and the exchange must not be presented as a real unprompted ChatGPT output endorsing the product. May propose a compliant angle (mechanism education grounded in documented facts) or steer to the Apple Notes confession format as the lower-risk fit for a transformation story. For the Notes version: title-as-hook, first-person list with the product as the least enthusiastic line, keyboard-taps-only audio, grounding realizations in real reviews. Should apply surface-selection guidance rather than treating the three formats as interchangeable.",
      "assertions": [
        "Applies the iOS-native reveal surfaces section of the imessage-video-ads reference",
        "Flags health-claim/substantiation risk for the fabricated ChatGPT answer before or while drafting",
        "Does not present the ChatGPT exchange as a real unprompted output endorsing the product",
        "Recommends legal review or a compliant reframe for health advice in the AI answer",
        "Apple Notes guidance: title-as-hook, first-person confession, product as an understated list item, keyboard-taps-only audio",
        "Grounds claims and realizations in documented facts/reviews (Grounded Inputs)",
        "Gives surface-selection reasoning (ChatGPT vs Notes) instead of treating formats as interchangeable"
      ],
      "files": []
    },
    {
      "id": 9,
      "prompt": "Our Meta account is stuck — we've tested 30 ads over two months and nothing beats the control. I have our reviews exported and access to our ad account data. Build me a creative plan for next month.",
      "expected_output": "Should apply Mode 4 / references/creative-roadmap.md rather than jumping straight to generating ads. Should identify the account as exploration state (nothing working) and shape the plan accordingly: mostly net-new concepts across different segments/angles, minimal iterations, per-metric win redefinition (a hold-rate lift or CPC drop counts as a hit worth pulling on). Should synthesize the three signals (account performance from the ad data, customer language from the reviews, external organic — asking for or mining niche organic content) into concepts ranked by evidence tier, each with a cited source. Should produce a capacity-checked monthly slate with production tiers (favoring T1/T2 low-fidelity tests per the fidelity ladder) and flag the common exploration-state root causes to check (boring creative, overcomplicated message, unclear UVP, punishing CPMs). Should end with the retro plan for judging the slate at month end. Should not invent customer language or claims — insights must trace to the provided reviews/data.",
      "assertions": [
        "Applies the creative strategy loop (Mode 4) instead of only generating ad copy",
        "Diagnoses exploration state and recommends a wide, net-new-heavy mix with minimal iterations",
        "Redefines wins per-metric for a stuck account",
        "Synthesizes all three signal sources or explicitly requests the missing one",
        "Concepts are evidence-ranked with cited sources (no invented insights)",
        "Monthly slate is capacity-checked and production-tiered, favoring low-fidelity tests",
        "Includes a month-end retro plan that feeds the next slate"
      ],
      "files": []
    },
    {
      "id": 10,
      "prompt": "We generated four ad concepts for a client (an organic skincare brand) and need to send them something they can actually look at and approve — with the Instagram preview, the carousel frames, and the different headline options they can compare. Can you put that together?",
      "expected_output": "Should recognize this as a creative review page request and apply references/creative-review-page.md + the assets/creative-review-template.html template rather than producing plain markdown. Should copy the template into the output folder and populate its DATA object with the four concepts as tabs, each with an in-feed Instagram preview, a labeled frame-by-frame storyboard (frames labeled by narrative job — Hook / Problem / Proof / Ask — not by pictured content), selectable headline variations, primary text, and destination/CTA. Should curate to a reviewable number of concepts (2-4) rather than dumping everything. Should include a required grounding disclosure per concept stating what is real (product photography, any claims/results) and label illustrative proof as illustrative — never present invented stats or stock imagery as the brand's own. Should use styled placeholders for frames not yet rendered to image, and keep image paths relative. Should explain how to deliver it (open locally, host on a static host, or hand off the file).",
      "assertions": [
        "Produces a creative review page from the HTML template, not plain markdown",
        "Populates the DATA object (concept tabs, in-feed preview, frame storyboard, headline variations, copy, destination)",
        "Labels storyboard frames by narrative job rather than by pictured content",
        "Includes a required grounding/disclosure line per concept; labels illustrative proof as illustrative",
        "Does not present invented stats or stock imagery as the brand's real assets",
        "Uses placeholders for unrendered frames and keeps image paths relative",
        "Explains how to deliver the page (open locally / host / hand off the file)"
      ],
      "files": []
    },
    {
      "id": 11,
      "prompt": "I want to make one of those AirDrop-style video ads — where a phone gets an incoming AirDrop and you tap accept. We sell a limited-run sneaker drop.",
      "expected_output": "Should apply the AirDrop surface in references/imessage-video-ads.md (the iOS-native reveal family), not treat it as a novel format. Should build the ad around the interaction: an incoming AirDrop card (translucent sheet, sender device name, a preview thumbnail, gray Decline / blue Accept) from the receiver's POV, with the Accept tap as the reveal beat and the transfer progress-ring as the signature motion. Should make the preview thumbnail earn the tap (the sneaker money-shot / the drop), cast a relatable human sender name rather than the brand, use the AirDrop swoosh sound (not iMessage tritones) with the Apple trade-dress note, and keep it short. Should apply the family grounding/disclosure rules (a dramatization of a share, not a real endorsement; claims substantiated). May note receiver-POV-by-default vs sender-POV-as-flex.",
      "assertions": [
        "Applies the AirDrop iOS-native-reveal surface, not a from-scratch format",
        "Builds around the incoming-AirDrop-card + accept-tap-as-reveal interaction (receiver POV)",
        "Preview thumbnail is treated as the hook that must earn the accept",
        "Casts a relatable human sender name, not the brand, on the incoming card",
        "Uses the AirDrop swoosh sound + Apple trade-dress note, not iMessage tritones",
        "Applies the family grounding/disclosure rules (dramatized share, substantiated claims, not a real endorsement)"
      ],
      "files": []
    }
  ]
}
