From 9835f75e8e78d2a6306c3f941b1cbcf7558a3b54 Mon Sep 17 00:00:00 2001 From: skillfactor-pipeline Date: Fri, 14 Aug 2026 19:11:44 +0200 Subject: [PATCH] feat: boom-operator skill package v0.1.0 --- PROVENANCE.md | 31 ++++ SKILL.md | 53 ++++++ manifest.json | 268 +++++++++++++++++++++++++++++ references/ai-skills.md | 362 ++++++++++++++++++++++++++++++++++++++++ references/market.md | 62 +++++++ references/profile.md | 23 +++ references/skills.md | 70 ++++++++ references/tasks.md | 32 ++++ references/tools.md | 38 +++++ 9 files changed, 939 insertions(+) create mode 100644 PROVENANCE.md create mode 100644 SKILL.md create mode 100644 manifest.json create mode 100644 references/ai-skills.md create mode 100644 references/market.md create mode 100644 references/profile.md create mode 100644 references/skills.md create mode 100644 references/tasks.md create mode 100644 references/tools.md diff --git a/PROVENANCE.md b/PROVENANCE.md new file mode 100644 index 0000000..4f3c1ab --- /dev/null +++ b/PROVENANCE.md @@ -0,0 +1,31 @@ +# Data provenance — boom-operator + +Where the content of this skill package comes from, counted by +content items (tasks, competences, tools, evidence entries, curated +knowledge). Rendered live by Gitea: + +```mermaid +%%{init: {'theme':'base','themeVariables':{'pie1':'#f9a825','pie2':'#1e88e5','pie3':'#ff355e','pie4':'#d97757','pie5':'#8e24aa','pieOuterStrokeWidth':'0px','pieSectionTextColor':'#fff'}}}%% +pie showData + title Content sources — boom-operator + "ESCO (occupation & competences)" : 28 + "O*NET (tasks & tools)" : 56 + "Job boards (market evidence)" : 34 + "Anthropic official Claude skills" : 5 + "External AI skill packs (mapped)" : 174 +``` + +| Source | Items | Share | Files | +|---|---|---|---| +| ESCO (occupation & competences) | 28 | 9.4 % | references/profile.md, references/skills.md | +| O*NET (tasks & tools) | 56 | 18.9 % | references/tasks.md, references/tools.md | +| Job boards (market evidence) | 34 | 11.4 % | references/market.md (full report) + "Market evidence" headline sections | +| Wikipedia & AI expert curation | 0 | 0.0 % | glossary, literature, usecases, intake, quality, evals/ | +| Anthropic official Claude skills | 5 | 1.7 % | references/ai-skills.md, section "anthropics/skills" (official Claude Code skills) | +| External AI skill packs (mapped) | 174 | 58.6 % | references/ai-skills.md (per-source attribution inside) | +| Stack Exchange practitioner Q&A (CC-BY-SA) | 0 | 0.0 % | references/practitioner-qa.md (per-entry attribution inside) | + +Licensing: O*NET (USDOL/ETA, CC BY 4.0) · ESCO (© European Union) · +job-ad evidence via official APIs (JSearch/Adzuna) · Wikipedia content +paraphrased with source URLs — never copied · external AI skills are +linked, not copied (Apache-2.0/MIT/source-available, see ai-skills.md). diff --git a/SKILL.md b/SKILL.md new file mode 100644 index 0000000..1fb459d --- /dev/null +++ b/SKILL.md @@ -0,0 +1,53 @@ +--- +name: boom-operator +description: "Occupational skill for the role 'boom operator' (also: TV boom operator, shotgun microphone operator, production sound mixer assistant, shotgun operator, boom holder, sound equipment technician). Use when the user asks for typical boom operator work such as: typical boom operator responsibilities" +--- + +# Boom Operator + +Boom operators set up and operate the boom microphone, either by hand, on an arm or on a moving platform. They make sure that every microphone is correctly stationed on set and in the best position to capture the dialogues. Boom operators are also responsible for the microphones on the actors clothing. + +## Core workflow + + +## How to use this skill + +- Read [references/profile.md](references/profile.md) for the occupation profile and scope. +- Consult [references/tasks.md](references/tasks.md) for the full task and activity inventory. +- Check [references/skills.md](references/skills.md) for essential vs. optional competences. +- Check [references/tools.md](references/tools.md) for the software commonly used in this role. +- See [references/ai-skills.md](references/ai-skills.md) — matched external AI agent skills (per-source attribution). + +## Key competences (essential) + +- acoustics +- adapt to type of media +- analyse a script +- audiovisual equipment +- consult with sound editor +- follow directions of the artistic director +- follow work schedule +- manage sound quality +- perform soundchecks +- program sound cues +- set up sound equipment +- study media sources +- use audio reproduction software +- use technical documentation +- work ergonomically + +## Hot technologies + +- Adobe Acrobat +- Adobe Creative Cloud software +- Adobe InDesign +- Adobe Photoshop +- Apple macOS +- Autodesk AutoCAD +- Facebook +- Git +- Linux +- Microsoft Excel + +--- +*Sources: ESCO v1.2.1 (http://data.europa.eu/esco/occupation/4d102082-8d43-4c59-81ab-08a7509f3c40), O*NET 30.3 (27-4014.00, manual nearest match). See manifest.json for licensing/attribution.* diff --git a/manifest.json b/manifest.json new file mode 100644 index 0000000..31608c7 --- /dev/null +++ b/manifest.json @@ -0,0 +1,268 @@ +{ + "name": "boom-operator", + "title": "boom operator", + "version": "0.1.0", + "layer": "core", + "language": "en", + "generated": "2026-07-07", + "ids": { + "esco_uri": "http://data.europa.eu/esco/occupation/4d102082-8d43-4c59-81ab-08a7509f3c40", + "esco_code": "3521.1.1", + "isco_group": "3521", + "onet_soc": "27-4014.00", + "crosswalk_match": "manual nearest via ISCO 3521 (Sound Engineering Technicians)" + }, + "sources": [ + { + "name": "ESCO", + "version": "1.2.1", + "url": "https://esco.ec.europa.eu/" + }, + { + "name": "O*NET", + "version": "30.3", + "url": "https://www.onetcenter.org/", + "license": "CC BY 4.0" + } + ], + "attribution": "This package includes information from the O*NET Database (v30.3) by the U.S. Department of Labor, Employment and Training Administration (USDOL/ETA), CC BY 4.0. skillfactor is not endorsed by USDOL/ETA. ESCO data (v1.2.1) (c) European Union, used per the ESCO download conditions: https://esco.ec.europa.eu/en/use-esco/download", + "counts": { + "tasks": 0, + "dwas": 0, + "skills_essential": 15, + "skills_optional": 12, + "software": 0 + }, + "enrichment_ai_skills": { + "generated": "2026-07-14", + "method": "deterministic mapping (ISCO prefix + title/competence keywords)", + "sources": { + "anthropics/skills": { + "repo": "https://github.com/anthropics/skills", + "commit": "f6656c1", + "license": "Apache-2.0; the document skills (docx/pdf/pptx/xlsx) are source-available \u2014 see the LICENSE.txt in the upstream skill folder", + "skills": 2 + }, + "google/skills": { + "repo": "https://github.com/google/skills", + "commit": "b15f327", + "license": "Apache-2.0", + "skills": 1 + }, + "veniceai/skills": { + "repo": "https://github.com/veniceai/skills", + "commit": "de089fa", + "license": "MIT", + "skills": 5 + }, + "ConardLi/garden-skills": { + "repo": "https://github.com/ConardLi/garden-skills", + "commit": "fbd6453", + "license": "MIT", + "skills": 1 + }, + "a5c-ai/babysitter": { + "repo": "https://github.com/a5c-ai/babysitter", + "commit": "44a5d58b", + "license": "MIT", + "skills": 5 + }, + "jeremylongshore/claude-code-plugins-plus-skills": { + "repo": "https://github.com/jeremylongshore/claude-code-plugins-plus-skills", + "commit": "e112938a", + "license": "MIT", + "skills": 12 + }, + "davepoon/buildwithclaude": { + "repo": "https://github.com/davepoon/buildwithclaude", + "commit": "3c94e0c", + "license": "MIT", + "skills": 3 + }, + "sanjay3290/ai-skills": { + "repo": "https://github.com/sanjay3290/ai-skills", + "commit": "3619692", + "license": "Apache-2.0", + "skills": 2 + }, + "Orchestra-Research/AI-research-SKILLs": { + "repo": "https://github.com/Orchestra-Research/AI-research-SKILLs", + "commit": "773a529", + "license": "MIT", + "skills": 2 + }, + "davila7/claude-code-templates": { + "repo": "https://github.com/davila7/claude-code-templates", + "commit": "fa79251", + "license": "MIT", + "skills": 4 + }, + "zechenzhangAGI/AI-research-SKILLs": { + "repo": "https://github.com/zechenzhangAGI/AI-research-SKILLs", + "commit": "773a529", + "license": "MIT", + "skills": 2 + }, + "EveryInc/compound-engineering-plugin": { + "repo": "https://github.com/EveryInc/compound-engineering-plugin", + "commit": "1a7a4c1", + "license": "MIT", + "skills": 1 + }, + "gooseworks-ai/goose-skills": { + "repo": "https://github.com/gooseworks-ai/goose-skills", + "commit": "94ec916", + "license": "no explicit license \u2014 referenced by link only", + "skills": 4 + }, + "nexu-io/open-design": { + "repo": "https://github.com/nexu-io/open-design", + "commit": "4b66023", + "license": "Apache-2.0", + "skills": 5 + }, + "NoizAI/skills": { + "repo": "https://github.com/NoizAI/skills", + "commit": "2a0e09d", + "license": "no explicit license \u2014 referenced by link only", + "skills": 4 + }, + "Vincentwei1021/video-shotcraft": { + "repo": "https://github.com/Vincentwei1021/video-shotcraft", + "commit": "d491544", + "license": "Apache-2.0", + "skills": 1 + }, + "bitwize-music-studio/claude-ai-music-skills": { + "repo": "https://github.com/bitwize-music-studio/claude-ai-music-skills", + "commit": "96446de", + "license": "custom (see upstream LICENSE)", + "skills": 3 + }, + "affaan-m/everything-claude-code": { + "repo": "https://github.com/affaan-m/everything-claude-code", + "commit": "ed38744", + "license": "MIT", + "skills": 1 + }, + "SamurAIGPT/Generative-Media-Skills": { + "repo": "https://github.com/SamurAIGPT/Generative-Media-Skills", + "commit": "a1c4c98", + "license": "MIT", + "skills": 2 + }, + "mukul975/Anthropic-Cybersecurity-Skills": { + "repo": "https://github.com/mukul975/Anthropic-Cybersecurity-Skills", + "commit": "673da1f", + "license": "Apache-2.0", + "skills": 2 + }, + "mohitagw15856/pm-claude-skills": { + "repo": "https://github.com/mohitagw15856/pm-claude-skills", + "commit": "876fa30", + "license": "MIT", + "skills": 2 + }, + "transloadit/skills": { + "repo": "https://github.com/transloadit/skills", + "commit": "8dd2fd9", + "license": "no explicit license \u2014 referenced by link only", + "skills": 1 + }, + "AgriciDaniel/claude-blog": { + "repo": "https://github.com/AgriciDaniel/claude-blog", + "commit": "49842ea", + "license": "MIT", + "skills": 1 + }, + "infrasity-labs/dev-gtm-claude-skills": { + "repo": "https://github.com/infrasity-labs/dev-gtm-claude-skills", + "commit": "02cfefb", + "license": "MIT", + "skills": 1 + }, + "silverstein/minutes": { + "repo": "https://github.com/silverstein/minutes", + "commit": "3fb2e83", + "license": "MIT", + "skills": 1 + }, + "K-Dense-AI/claude-scientific-skills": { + "repo": "https://github.com/K-Dense-AI/claude-scientific-skills", + "commit": "4d97e29", + "license": "MIT", + "skills": 1 + }, + "K-Dense-AI/scientific-agent-skills": { + "repo": "https://github.com/K-Dense-AI/scientific-agent-skills", + "commit": "4d97e29", + "license": "MIT", + "skills": 1 + }, + "Orkas-AI/Orkas-VideoStudio": { + "repo": "https://github.com/Orkas-AI/Orkas-VideoStudio", + "commit": "dd4a0f4", + "license": "MIT", + "skills": 1 + }, + "disler/claude-code-hooks-multi-agent-observability": { + "repo": "https://github.com/disler/claude-code-hooks-multi-agent-observability", + "commit": "8a6e5cf", + "license": "no explicit license \u2014 referenced by link only", + "skills": 1 + }, + "microsoft/skills": { + "repo": "https://github.com/microsoft/skills", + "commit": "dc543aa", + "license": "MIT", + "skills": 5 + }, + "indranilbanerjee/digital-marketing-pro": { + "repo": "https://github.com/indranilbanerjee/digital-marketing-pro", + "commit": "a3d119c", + "license": "MIT", + "skills": 1 + }, + "malob/nix-config": { + "repo": "https://github.com/malob/nix-config", + "commit": "f2ed178", + "license": "MIT", + "skills": 1 + }, + "ljagiello/ctf-skills": { + "repo": "https://github.com/ljagiello/ctf-skills", + "commit": "d19f35f", + "license": "MIT", + "skills": 1 + } + }, + "total_skills": 80, + "tiers": { + "core": 15, + "adjacent": 65 + } + }, + "provenance": { + "items": { + "esco": 28, + "onet": 56, + "jobads": 34, + "wiki_ai": 0, + "anthropic": 5, + "ai_skills": 174, + "stackx": 0 + }, + "share_percent": { + "esco": 9.4, + "onet": 18.9, + "jobads": 11.4, + "wiki_ai": 0.0, + "anthropic": 1.7, + "ai_skills": 58.6, + "stackx": 0.0 + }, + "method": "content items per source category" + }, + "collar": "white", + "computer_work": true +} \ No newline at end of file diff --git a/references/ai-skills.md b/references/ai-skills.md new file mode 100644 index 0000000..ea1a1a8 --- /dev/null +++ b/references/ai-skills.md @@ -0,0 +1,362 @@ +# External AI agent skills — boom-operator + +Proven, publicly available AI agent skills mapped to this occupation. +Nothing is copied from the sources: every entry is a name, a one-line +summary and a link to the upstream skill package. Each section names +its source repository, commit, license and retrieval date. + +**Tiers:** `core` = the skill directly exercises a top market hard +skill, tool or method (from gated job-ad evidence) or an essential +ESCO competence of this occupation; `adjacent` = +plausibly useful, secondary. Entries are capped at 12 per source +and 80 in total per occupation (core first, +strongest matches survive); everything beyond the caps is excluded +and logged in the pipeline audit trail, not in this package. + +_Matched deterministically (ISCO group + title/competence keywords, +tiered against market evidence + ESCO essentials) by +`pipeline/p5_enrich_ai_skills.py` on 2026-07-14._ + +## Source: anthropics/skills + +- Repository: [https://github.com/anthropics/skills](https://github.com/anthropics/skills) (commit `f6656c1`, retrieved 2026-07-14) +- License: Apache-2.0; the document skills (docx/pdf/pptx/xlsx) are source-available — see the LICENSE.txt in the upstream skill folder + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `docx` | adjacent | Use this skill whenever the user wants to create, read, edit, or manipulate Word documents (.docx files) or Word templates (.dotx files). Triggers include: any mention of 'Word doc', 'word document', '.docx', '.dotx', or requests to … | [source](https://github.com/anthropics/skills/tree/main/skills/docx) | +| `pdf` | adjacent | Use this skill whenever the user wants to do anything with PDF files. This includes reading or extracting text/tables from PDFs, combining or merging multiple PDFs into one, splitting PDFs apart, rotating pages, adding watermarks, creating … | [source](https://github.com/anthropics/skills/tree/main/skills/pdf) | + +## Source: ConardLi/garden-skills + +- Repository: [https://github.com/ConardLi/garden-skills](https://github.com/ConardLi/garden-skills) (commit `fbd6453`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `web-video-presentation` | adjacent | 把一篇文章或口播稿,做成"看起来像视频"的点击驱动 16:9 网页演示,可选合成口播音频。流程:原始文章 → **一次产出**口播稿 + outline 开发计划 → 用户**一次对齐** 5 件事(稿子 / outline / 主题 / 素材 / 开发模式)→ 网页开发(逐章 / 顺序 / 并行)→ 可选音频合成(provider-agnostic:内置 MiniMax mmx-cli + OpenAI TTS,可换 ElevenLabs / edge-tts / … | [source](https://github.com/ConardLi/garden-skills/tree/fbd6453/skills/web-video-presentation) | + +## Source: a5c-ai/babysitter + +- Repository: [https://github.com/a5c-ai/babysitter](https://github.com/a5c-ai/babysitter) (commit `44a5d58b`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `sound-design-direction` | core | Create comprehensive audio design including music cues, sound effects, Foley, and score direction | [source](https://github.com/a5c-ai/babysitter/tree/44a5d58b/library/specializations/domains/social-sciences-humanities/arts-culture/film-tv-production/skills/sound-design-direction) | +| `procedural-audio` | adjacent | Procedural sound skill for synthesis and dynamic sound design. | [source](https://github.com/a5c-ai/babysitter/tree/44a5d58b/library/specializations/game-development/skills/procedural-audio) | +| `ipa-transcription-phonological` | adjacent | Transcribe speech using International Phonetic Alphabet and analyze sound systems including phonotactics and phonological rules | [source](https://github.com/a5c-ai/babysitter/tree/44a5d58b/library/specializations/domains/social-sciences-humanities/humanities/skills/ipa-transcription-phonological) | +| `wwise` | adjacent | Wwise integration skill for sound banks, RTPC, and interactive music. | [source](https://github.com/a5c-ai/babysitter/tree/44a5d58b/library/specializations/game-development/skills/wwise) | +| `audio-dsp` | adjacent | Audio DSP skill for filters and real-time processing. | [source](https://github.com/a5c-ai/babysitter/tree/44a5d58b/library/specializations/game-development/skills/audio-dsp) | + +## Source: affaan-m/everything-claude-code + +- Repository: [https://github.com/affaan-m/everything-claude-code](https://github.com/affaan-m/everything-claude-code) (commit `ed38744`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `fal-ai-media` | adjacent | Unified media generation via fal.ai MCP — image, video, and audio. Covers text-to-image (Nano Banana), text/image-to-video (Seedance, Kling, Veo 3), text-to-speech (CSM-1B), and video-to-audio (ThinkSound). Use when the user wants to … | [source](https://github.com/affaan-m/everything-claude-code/tree/ed38744/.agents/skills/fal-ai-media) | + +## Source: AgriciDaniel/claude-blog + +- Repository: [https://github.com/AgriciDaniel/claude-blog](https://github.com/AgriciDaniel/claude-blog) (commit `49842ea`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `blog-audio` | adjacent | Generate audio narration of blog posts using Google Gemini TTS. Supports summary narration, full article read-aloud, and two-speaker podcast/dialogue mode with 30 voice options. Outputs MP3 with HTML5 audio embed code. Works standalone via … | [source](https://github.com/AgriciDaniel/claude-blog/tree/49842ea/skills/blog-audio) | + +## Source: bitwize-music-studio/claude-ai-music-skills + +- Repository: [https://github.com/bitwize-music-studio/claude-ai-music-skills](https://github.com/bitwize-music-studio/claude-ai-music-skills) (commit `96446de`, retrieved 2026-07-14) +- License: custom (see upstream LICENSE) + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `promo-director` | adjacent | Generates 15-second vertical promo videos for social media from mastered audio. Use after mastering is complete and before release, when the user wants social media content. | [source](https://github.com/bitwize-music-studio/claude-ai-music-skills/tree/96446de/skills/promo-director) | +| `import-audio` | adjacent | Moves audio files to the correct album location with proper path structure. Use when the user has downloaded WAV files from Suno or other sources that need to be organized. | [source](https://github.com/bitwize-music-studio/claude-ai-music-skills/tree/96446de/skills/import-audio) | +| `import-art` | adjacent | Places album art files in the correct audio and content directory locations. Use when the user has generated or downloaded album artwork that needs to be saved. | [source](https://github.com/bitwize-music-studio/claude-ai-music-skills/tree/96446de/skills/import-art) | + +## Source: davepoon/buildwithclaude + +- Repository: [https://github.com/davepoon/buildwithclaude](https://github.com/davepoon/buildwithclaude) (commit `3c94e0c`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `resemble-detect` | core | Deepfake detection and media safety — detect AI-generated audio, images, video, and text, trace synthesis sources, apply watermarks, verify speaker identity, and analyze media intelligence using Resemble AI | [source](https://github.com/davepoon/buildwithclaude/tree/3c94e0c/plugins/all-skills/skills/resemble-detect) | +| `tubeify` | core | Remove pauses, filler words (um, uh), and dead air from raw YouTube recordings via the Tubeify API. Use when the user wants to edit a video, clean up audio, trim silences, or polish a raw recording for YouTube. | [source](https://github.com/davepoon/buildwithclaude/tree/3c94e0c/plugins/all-skills/skills/tubeify) | +| `image-enhancer` | adjacent | Improves the quality of images, especially screenshots, by enhancing resolution, sharpness, and clarity. Perfect for preparing images for presentations, documentation, or social media posts. | [source](https://github.com/davepoon/buildwithclaude/tree/3c94e0c/plugins/all-skills/skills/image-enhancer) | + +## Source: davila7/claude-code-templates + +- Repository: [https://github.com/davila7/claude-code-templates](https://github.com/davila7/claude-code-templates) (commit `fa79251`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `nemo-curator` | core | GPU-accelerated data curation for LLM training. Supports text/image/video/audio. Features fuzzy deduplication (16× faster), quality filtering (30+ heuristics), semantic deduplication, PII redaction, NSFW detection. Scales across GPUs with … | [source](https://github.com/davila7/claude-code-templates/tree/fa79251/cli-tool/components/skills/ai-research/data-processing-nemo-curator) | +| `audiocraft-audio-generation` | adjacent | PyTorch library for audio generation including text-to-music (MusicGen) and text-to-sound (AudioGen). Use when you need to generate music from text descriptions, create sound effects, or perform melody-conditioned music generation. | [source](https://github.com/davila7/claude-code-templates/tree/fa79251/cli-tool/components/skills/ai-research/multimodal-audiocraft) | +| `game-audio` | adjacent | Game audio principles. Sound design, music integration, adaptive audio systems. | [source](https://github.com/davila7/claude-code-templates/tree/fa79251/cli-tool/components/skills/creative-design/game-development/game-audio) | +| `image-enhancer` | adjacent | Improves the quality of images, especially screenshots, by enhancing resolution, sharpness, and clarity. Perfect for preparing images for presentations, documentation, or social media posts. | [source](https://github.com/davila7/claude-code-templates/tree/fa79251/cli-tool/components/skills/media/image-enhancer) | + +## Source: disler/claude-code-hooks-multi-agent-observability + +- Repository: [https://github.com/disler/claude-code-hooks-multi-agent-observability](https://github.com/disler/claude-code-hooks-multi-agent-observability) (commit `8a6e5cf`, retrieved 2026-07-14) +- License: no explicit license — referenced by link only + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `Video Processor` | adjacent | Process video files with audio extraction, format conversion (mp4, webm), and Whisper transcription. Use when user mentions video conversion, audio extraction, transcription, mp4, webm, ffmpeg, or whisper transcription. | [source](https://github.com/disler/claude-code-hooks-multi-agent-observability/tree/8a6e5cf/.claude/skills/video-processor) | + +## Source: EveryInc/compound-engineering-plugin + +- Repository: [https://github.com/EveryInc/compound-engineering-plugin](https://github.com/EveryInc/compound-engineering-plugin) (commit `1a7a4c1`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `ce-riffrec-feedback-analysis` | core | Analyze Riffrec feedback captures from bundles or standalone recordings. Always load for `riffrec-*.zip`, `session.json` + `events.json` + `recording.webm` + `voice.webm` bundles, `.mp4`/`.mov`/`.webm` videos, `.m4a`/`.mp3`/`.wav` audio, … | [source](https://github.com/EveryInc/compound-engineering-plugin/tree/1a7a4c1/skills/ce-riffrec-feedback-analysis) | + +## Source: gooseworks-ai/goose-skills + +- Repository: [https://github.com/gooseworks-ai/goose-skills](https://github.com/gooseworks-ai/goose-skills) (commit `94ec916`, retrieved 2026-07-14) +- License: no explicit license — referenced by link only + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `video-polish` | core | Takes an existing screen recording or demo video and adds professional zoom/pan effects synchronized to the narration. Uses transcript-driven zoom targeting and Remotion for rendering. Optionally replaces audio with a soundtrack. | [source](https://github.com/gooseworks-ai/goose-skills/tree/94ec916/skills/design/packs/video-production/video-polish) | +| `review-ugc-render` | adjacent | Mandatory pre-publish review gate for a UGC video render. Transcribes the finished render's AUDIO with Whisper and word-diffs it against the approved spoken script, then gates set_final_render — blocking a render whose generated audio … | [source](https://github.com/gooseworks-ai/goose-skills/tree/94ec916/skills/ads/packs/ugc-video-formats/review-ugc-render) | +| `beat-sync-reel` | adjacent | Generates Instagram Reels where product image cuts are synced to audio beats. Accepts audio as a local file, URL, or search query. Uses librosa for beat detection, FFmpeg Ken Burns for scene animation, and Pillow for text overlays. No AI … | [source](https://github.com/gooseworks-ai/goose-skills/tree/94ec916/skills/design/packs/video-production/beat-sync-reel) | +| `create-video-seedance-2-fal` | adjacent | Generate a single 4-15s vertical video clip with ByteDance Seedance 2.0 reference-to-video via fal.ai. Multi-image reference (avatar + product + setting), native lip-synced VO + ambient audio via `generate_audio: true`, internal multi-cut … | [source](https://github.com/gooseworks-ai/goose-skills/tree/94ec916/skills/ads/packs/ugc-video-formats/create-video-seedance-2-fal) | + +## Source: indranilbanerjee/digital-marketing-pro + +- Repository: [https://github.com/indranilbanerjee/digital-marketing-pro](https://github.com/indranilbanerjee/digital-marketing-pro) (commit `a3d119c`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `c2pa-metadata` | adjacent | Embed C2PA (Content Authenticity Initiative) provenance manifests in AI-generated marketing assets (image/video/audio/PDF). Use when: preparing AI-generated ad creative, social images, or video for EU markets to comply with EU AI Act … | [source](https://github.com/indranilbanerjee/digital-marketing-pro/tree/a3d119c/skills/c2pa-metadata) | + +## Source: infrasity-labs/dev-gtm-claude-skills + +- Repository: [https://github.com/infrasity-labs/dev-gtm-claude-skills](https://github.com/infrasity-labs/dev-gtm-claude-skills) (commit `02cfefb`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `blog-audio` | adjacent | Generate audio narration of blog posts using Google Gemini TTS. Supports summary narration, full article read-aloud, and two-speaker podcast/dialogue mode with 30 voice options. Outputs MP3 with HTML5 audio embed code. Works standalone via … | [source](https://github.com/infrasity-labs/dev-gtm-claude-skills/tree/02cfefb/.claude/skills/blog-audio) | + +## Source: jeremylongshore/claude-code-plugins-plus-skills + +- Repository: [https://github.com/jeremylongshore/claude-code-plugins-plus-skills](https://github.com/jeremylongshore/claude-code-plugins-plus-skills) (commit `e112938a`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `abridge-core-workflow-a` | core | Implement Abridge ambient clinical documentation capture-to-note pipeline. Use when building the primary encounter workflow: audio capture, real-time transcription, AI note generation, and EHR note insertion. Trigger: "abridge clinical … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/abridge-pack/skills/abridge-core-workflow-a) | +| `granola-common-errors` | core | Troubleshoot common Granola errors \u2014 audio capture failures, transcription\ \ issues,\ncalendar sync problems, and integration errors. Platform-specific fixes\ \ for macOS and Windows.\nTrigger: \"granola error\", \"granola not … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/granola-pack/skills/granola-common-errors) | +| `granola-performance-tuning` | core | Optimize Granola transcription accuracy, note quality, and processing speed. Use when improving transcription quality, reducing processing time, optimizing templates for better AI output, or tuning audio setup. Trigger: "granola … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/granola-pack/skills/granola-performance-tuning) | +| `twinmind-performance-tuning` | core | Optimize TwinMind transcription accuracy and speed with Ear-3 model configuration, audio quality tuning, and caching strategies. Use when implementing performance tuning, or managing TwinMind meeting AI operations. Trigger with phrases … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/twinmind-pack/skills/twinmind-performance-tuning) | +| `granola-install-auth` | core | Install and configure Granola AI meeting notes with calendar and audio permissions. Use when setting up Granola for the first time, connecting Google/Outlook calendars, granting macOS Screen Recording permission, or configuring Windows … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/granola-pack/skills/granola-install-auth) | +| `elevenlabs-core-workflow-b` | adjacent | Implement ElevenLabs speech-to-speech, sound effects, audio isolation, and speech-to-text. Use when converting voice to another voice, generating sound effects from text, removing background noise, or transcribing audio. Trigger: … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/elevenlabs-pack/skills/elevenlabs-core-workflow-b) | +| `twinmind-security-basics` | adjacent | Security best practices for TwinMind: on-device audio processing, encrypted cloud backups, microphone permissions, and data privacy controls. Use when implementing security basics, or managing TwinMind meeting AI operations. Trigger with … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/twinmind-pack/skills/twinmind-security-basics) | +| `abridge-common-errors` | adjacent | Diagnose and fix common Abridge clinical AI integration errors. Use when encountering EHR connectivity failures, note generation errors, audio streaming issues, or FHIR validation problems with Abridge. Trigger: "abridge error", "abridge … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/abridge-pack/skills/abridge-common-errors) | +| `abridge-performance-tuning` | adjacent | Optimize Abridge clinical AI integration performance for high-volume deployments. Use when reducing note generation latency, optimizing audio streaming throughput, improving FHIR push performance, or scaling for multi-site health systems. … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/abridge-pack/skills/abridge-performance-tuning) | +| `assemblyai-core-workflow-a` | adjacent | Execute AssemblyAI primary workflow: async transcription with audio intelligence. Use when transcribing audio/video files, enabling speaker diarization, sentiment analysis, entity detection, PII redaction, or content moderation. Trigger … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/assemblyai-pack/skills/assemblyai-core-workflow-a) | +| `assemblyai-core-workflow-b` | adjacent | Execute AssemblyAI streaming transcription and LeMUR workflows. Use when implementing real-time speech-to-text, live captions, voice agents, or LLM-powered audio analysis with LeMUR. Trigger with phrases like "assemblyai streaming", … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/assemblyai-pack/skills/assemblyai-core-workflow-b) | +| `deepgram-core-workflow-a` | adjacent | Implement production pre-recorded speech-to-text with Deepgram. Use when building audio transcription, batch processing, or implementing diarization and intelligence features. Trigger: "deepgram transcription", "speech to text", … | [source](https://github.com/jeremylongshore/claude-code-plugins-plus-skills/tree/e112938a/plugins/saas-packs/deepgram-pack/skills/deepgram-core-workflow-a) | + +## Source: K-Dense-AI/claude-scientific-skills + +- Repository: [https://github.com/K-Dense-AI/claude-scientific-skills](https://github.com/K-Dense-AI/claude-scientific-skills) (commit `4d97e29`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `open-notebook` | adjacent | Self-hosted, open-source alternative to Google NotebookLM for AI-powered research and document analysis. Use when organizing research materials into notebooks, ingesting diverse content sources (PDFs, videos, audio, web pages, Office … | [source](https://github.com/K-Dense-AI/claude-scientific-skills/tree/4d97e29/skills/open-notebook) | + +## Source: K-Dense-AI/scientific-agent-skills + +- Repository: [https://github.com/K-Dense-AI/scientific-agent-skills](https://github.com/K-Dense-AI/scientific-agent-skills) (commit `4d97e29`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `open-notebook` | adjacent | Self-hosted, open-source alternative to Google NotebookLM for AI-powered research and document analysis. Use when organizing research materials into notebooks, ingesting diverse content sources (PDFs, videos, audio, web pages, Office … | [source](https://github.com/K-Dense-AI/scientific-agent-skills/tree/4d97e29/skills/open-notebook) | + +## Source: ljagiello/ctf-skills + +- Repository: [https://github.com/ljagiello/ctf-skills](https://github.com/ljagiello/ctf-skills) (commit `d19f35f`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `ctf-misc` | adjacent | Provides miscellaneous CTF challenge techniques for problems that do not cleanly fit the main categories. Use for encoding puzzles, pyjails, bash jails, RF/SDR, DNS oddities, unicode tricks, esoteric languages, QR or audio puzzles, … | [source](https://github.com/ljagiello/ctf-skills/tree/d19f35f/ctf-misc) | + +## Source: malob/nix-config + +- Repository: [https://github.com/malob/nix-config](https://github.com/malob/nix-config) (commit `f2ed178`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `cancel` | adjacent | Stop any currently playing text-to-speech audio | [source](https://github.com/malob/nix-config/tree/f2ed178/configs/claude/plugins/tts/skills/cancel) | + +## Source: microsoft/skills + +- Repository: [https://github.com/microsoft/skills](https://github.com/microsoft/skills) (commit `dc543aa`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `azure-ai-contentunderstanding-py` | adjacent | Azure AI Content Understanding SDK for Python. Use for multimodal content extraction from documents, images, audio, and video. Triggers: "azure-ai-contentunderstanding", "ContentUnderstandingClient", "multimodal analysis", "document … | [source](https://github.com/microsoft/skills/tree/dc543aa/.github/plugins/azure-sdk-python/skills/azure-ai-contentunderstanding-py) | +| `azure-ai-openai-dotnet` | adjacent | Azure OpenAI SDK for .NET. Client library for Azure OpenAI and OpenAI services. Use for chat completions, embeddings, image generation, audio transcription, and assistants. Triggers: "Azure OpenAI", "AzureOpenAIClient", "ChatClient", "chat … | [source](https://github.com/microsoft/skills/tree/dc543aa/.github/plugins/azure-sdk-dotnet/skills/azure-ai-openai-dotnet) | +| `azure-ai-voicelive-java` | adjacent | Azure AI VoiceLive SDK for Java. Real-time bidirectional voice conversations with AI assistants using WebSocket. Triggers: "VoiceLiveClient java", "voice assistant java", "real-time voice java", "audio streaming java", "voice activity … | [source](https://github.com/microsoft/skills/tree/dc543aa/.github/plugins/azure-sdk-java/skills/azure-ai-voicelive-java) | +| `azure-ai-voicelive-py` | adjacent | Build real-time voice AI applications using Azure AI Voice Live SDK (azure-ai-voicelive). Use this skill when creating Python applications that need real-time bidirectional audio communication with Azure AI, including voice assistants, … | [source](https://github.com/microsoft/skills/tree/dc543aa/.github/plugins/azure-sdk-python/skills/azure-ai-voicelive-py) | +| `azure-speech-to-text-rest-py` | adjacent | Azure Speech to Text REST API for short audio (Python). Use for simple speech recognition of audio files up to 60 seconds without the Speech SDK. Triggers: "speech to text REST", "short audio transcription", "speech recognition REST API", … | [source](https://github.com/microsoft/skills/tree/dc543aa/.github/plugins/azure-sdk-python/skills/azure-speech-to-text-rest-py) | + +## Source: mohitagw15856/pm-claude-skills + +- Repository: [https://github.com/mohitagw15856/pm-claude-skills](https://github.com/mohitagw15856/pm-claude-skills) (commit `876fa30`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `style-fingerprint` | adjacent | Study 3-5 documents the user actually shipped and distil a compact style card — so every skill writes in their voice, not the model's. Use when asked to learn my writing style, make outputs sound like me, build a voice profile, or when a … | [source](https://github.com/mohitagw15856/pm-claude-skills/tree/876fa30/plugins/pm-essentials/skills/style-fingerprint) | +| `youtube-script-writer` | adjacent | Write engaging, high-retention YouTube video scripts with visual and audio cues. Use when asked to write a YouTube script, design a video outline, draft a video hook, or structure a video narrative. Produces a polished script with multiple … | [source](https://github.com/mohitagw15856/pm-claude-skills/tree/876fa30/plugins/pm-writers/skills/youtube-script-writer) | + +## Source: mukul975/Anthropic-Cybersecurity-Skills + +- Repository: [https://github.com/mukul975/Anthropic-Cybersecurity-Skills](https://github.com/mukul975/Anthropic-Cybersecurity-Skills) (commit `673da1f`, retrieved 2026-07-14) +- License: Apache-2.0 + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `performing-steganography-detection` | adjacent | Detect and extract hidden data embedded in images, audio, and other media files using steganalysis tools to uncover covert communication channels. | [source](https://github.com/mukul975/Anthropic-Cybersecurity-Skills/tree/673da1f/skills/performing-steganography-detection) | +| `detecting-deepfake-audio-in-vishing-attacks` | adjacent | Detects AI-generated deepfake audio used in voice phishing (vishing) attacks by extracting spectral features (MFCC, spectral centroid, spectral contrast, zero-crossing rate) and classifying samples with machine learning models. Supports … | [source](https://github.com/mukul975/Anthropic-Cybersecurity-Skills/tree/673da1f/skills/detecting-deepfake-audio-in-vishing-attacks) | + +## Source: nexu-io/open-design + +- Repository: [https://github.com/nexu-io/open-design](https://github.com/nexu-io/open-design) (commit `4b66023`, retrieved 2026-07-14) +- License: Apache-2.0 + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `audio-jingle` | adjacent | Audio generation skill — jingles, beds, voiceover, and sound effects. Routes music requests to Suno V5 / Udio / Lyria, speech to MiniMax TTS / FishAudio / ElevenLabs V3, and SFX to ElevenLabs SFX or AudioCraft. Output is one MP3/WAV file … | [source](https://github.com/nexu-io/open-design/tree/4b66023/design-templates/audio-jingle) | +| `od-media-generation` | adjacent | Default reference pipeline for image, video, and audio projects — routes through media-image / media-video / media-audio atoms based on the project kind, wraps the output in a live artifact, and devloops on critique-theater until the score … | [source](https://github.com/nexu-io/open-design/tree/4b66023/plugins/_official/scenarios/od-media-generation) | +| `fal-lip-sync` | adjacent | Create talking head videos and lip sync audio to video via fal.ai. Useful for explainer avatars, multilingual dubbing previews, and social cuts. | [source](https://github.com/nexu-io/open-design/tree/4b66023/skills/fal-lip-sync) | +| `fal-video-edit` | adjacent | Edit existing videos using AI — remix style, upscale, remove background, and add audio via fal.ai's hosted video models. | [source](https://github.com/nexu-io/open-design/tree/4b66023/skills/fal-video-edit) | +| `hyperframes` | adjacent | Create video compositions, animations, title cards, overlays, captions, voiceovers, audio-reactive visuals, and scene transitions in HyperFrames HTML. Use when asked to build any HTML-based video content, add captions or subtitles synced … | [source](https://github.com/nexu-io/open-design/tree/4b66023/design-templates/hyperframes) | + +## Source: NoizAI/skills + +- Repository: [https://github.com/NoizAI/skills](https://github.com/NoizAI/skills) (commit `2a0e09d`, retrieved 2026-07-14) +- License: no explicit license — referenced by link only + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `sound-fx` | adjacent | Use this skill whenever the user wants to generate sound effects, ambient audio, or short audio clips from a text description. Triggers include: any mention of 'sound effect', 'sfx', 'generate sound', 'make a sound', 'audio effect', … | [source](https://github.com/NoizAI/skills/tree/2a0e09d/skills/sound-fx) | +| `characteristic-voice` | adjacent | Use this skill whenever the user wants speech to sound more human, companion-like, or emotionally expressive. Triggers include: any mention of 'say like', 'talk like', 'speak like', 'companion voice', 'comfort me', 'cheer me up', 'sound … | [source](https://github.com/NoizAI/skills/tree/2a0e09d/skills/characteristic-voice) | +| `daily-news-caster` | adjacent | Fetches the latest news using news-aggregator-skill, formats it into a podcast script in Markdown format, and uses the tts skill to generate a podcast audio file. Use when the user asks to get the latest news and read it out as a podcast. | [source](https://github.com/NoizAI/skills/tree/2a0e09d/skills/daily-news-caster) | +| `chat-with-anyone` | adjacent | Chat with any real person or fictional character in their own voice by automatically finding their speech online, extracting a clean reference sample, and generating audio replies. Also supports generating a matching voice from an uploaded … | [source](https://github.com/NoizAI/skills/tree/2a0e09d/skills/chat-with-anyone) | + +## Source: Orchestra-Research/AI-research-SKILLs + +- Repository: [https://github.com/Orchestra-Research/AI-research-SKILLs](https://github.com/Orchestra-Research/AI-research-SKILLs) (commit `773a529`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `nemo-curator` | core | GPU-accelerated data curation for LLM training. Supports text/image/video/audio. Features fuzzy deduplication (16× faster), quality filtering (30+ heuristics), semantic deduplication, PII redaction, NSFW detection. Scales across GPUs with … | [source](https://github.com/Orchestra-Research/AI-research-SKILLs/tree/773a529/05-data-processing/nemo-curator) | +| `audiocraft-audio-generation` | adjacent | PyTorch library for audio generation including text-to-music (MusicGen) and text-to-sound (AudioGen). Use when you need to generate music from text descriptions, create sound effects, or perform melody-conditioned music generation. | [source](https://github.com/Orchestra-Research/AI-research-SKILLs/tree/773a529/18-multimodal/audiocraft) | + +## Source: Orkas-AI/Orkas-VideoStudio + +- Repository: [https://github.com/Orkas-AI/Orkas-VideoStudio](https://github.com/Orkas-AI/Orkas-VideoStudio) (commit `dd4a0f4`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `video-craft` | adjacent | The craft standard that makes a video GOOD, not just rendered — hooks, story pacing, visual hierarchy, type & safe zones, motion/easing, captions, audio mix, platform conventions, shot language, generation-prompt writing, and a pre-publish … | [source](https://github.com/Orkas-AI/Orkas-VideoStudio/tree/dd4a0f4/packages/skills/video-craft) | + +## Source: SamurAIGPT/Generative-Media-Skills + +- Repository: [https://github.com/SamurAIGPT/Generative-Media-Skills](https://github.com/SamurAIGPT/Generative-Media-Skills) (commit `a1c4c98`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `muapi-media-generation` | adjacent | Generate AI images, videos, music, and audio from the terminal via muapi.ai — supports 100+ models including Flux, Midjourney v7, Kling 3.0, Veo3, and Suno V5 | [source](https://github.com/SamurAIGPT/Generative-Media-Skills/tree/a1c4c98/core/media) | +| `muapi-ugc-video-factory` | adjacent | Turn a person photo + a product photo + an optional script into a vertical 9:16 UGC-style video ad. Generates a lifestyle hero image (Nano-Banana Pro Edit), then animates it with native audio using Seedance 2.0 VIP image-to-video. | [source](https://github.com/SamurAIGPT/Generative-Media-Skills/tree/a1c4c98/.opencode/skills/muapi-ugc-video-factory) | + +## Source: sanjay3290/ai-skills + +- Repository: [https://github.com/sanjay3290/ai-skills](https://github.com/sanjay3290/ai-skills) (commit `3619692`, retrieved 2026-07-14) +- License: Apache-2.0 + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `google-tts` | core | Convert documents and text to audio using Google Cloud Text-to-Speech. Use this skill when the user wants to: narrate a document, read aloud text, generate audio from a file, convert text to speech, create a recording of documentation or … | [source](https://github.com/sanjay3290/ai-skills/tree/3619692/skills/google-tts) | +| `elevenlabs` | adjacent | Convert documents and text to audio using ElevenLabs text-to-speech. Use this skill when the user wants to create a podcast, narrate a document, read aloud text, generate audio from a file, or convert text to speech. | [source](https://github.com/sanjay3290/ai-skills/tree/3619692/skills/elevenlabs) | + +## Source: silverstein/minutes + +- Repository: [https://github.com/silverstein/minutes](https://github.com/silverstein/minutes) (commit `3fb2e83`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `minutes-setup` | adjacent | Guided first-time setup for Minutes — download whisper model, create directories, configure audio input. Use when the user says "set up minutes", "install minutes", "first time setup", "configure minutes", "get started with minutes", "how … | [source](https://github.com/silverstein/minutes/tree/3fb2e83/.agents/skills/minutes/minutes-setup) | + +## Source: transloadit/skills + +- Repository: [https://github.com/transloadit/skills](https://github.com/transloadit/skills) (commit `8dd2fd9`, retrieved 2026-07-14) +- License: no explicit license — referenced by link only + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `transform-transcribe-audio-with-transloadit` | adjacent | One-off transcription of local audio or video files to text or subtitle files using Transloadit via the official `@transloadit/node` CLI. Use when the user wants speech in local media converted to `.txt`, `.json`, `.srt`, or `.webvtt`; … | [source](https://github.com/transloadit/skills/tree/8dd2fd9/skills/transform-transcribe-audio-with-transloadit) | + +## Source: Vincentwei1021/video-shotcraft + +- Repository: [https://github.com/Vincentwei1021/video-shotcraft](https://github.com/Vincentwei1021/video-shotcraft) (commit `d491544`, retrieved 2026-07-14) +- License: Apache-2.0 + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `video-shotcraft` | adjacent | Create cinematic product videos from shot recipe cards, a validated template, and code/audio assets (Remotion + real page screenshots + 2.5D camera moves + beat-synced cuts + sound design). Use when the user asks to turn a frontend project … | [source](https://github.com/Vincentwei1021/video-shotcraft/tree/d491544) | + +## Source: zechenzhangAGI/AI-research-SKILLs + +- Repository: [https://github.com/zechenzhangAGI/AI-research-SKILLs](https://github.com/zechenzhangAGI/AI-research-SKILLs) (commit `773a529`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `nemo-curator` | core | GPU-accelerated data curation for LLM training. Supports text/image/video/audio. Features fuzzy deduplication (16× faster), quality filtering (30+ heuristics), semantic deduplication, PII redaction, NSFW detection. Scales across GPUs with … | [source](https://github.com/zechenzhangAGI/AI-research-SKILLs/tree/773a529/05-data-processing/nemo-curator) | +| `audiocraft-audio-generation` | adjacent | PyTorch library for audio generation including text-to-music (MusicGen) and text-to-sound (AudioGen). Use when you need to generate music from text descriptions, create sound effects, or perform melody-conditioned music generation. | [source](https://github.com/zechenzhangAGI/AI-research-SKILLs/tree/773a529/18-multimodal/audiocraft) | + +## Source: google/skills + +- Repository: [https://github.com/google/skills](https://github.com/google/skills) (commit `b15f327`, retrieved 2026-07-14) +- License: Apache-2.0 + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `ima-sdk-basics` | adjacent | Use this skill for Interactive Media Ads (IMA) SDK client-side ad insertion when you are requesting video ads client-side into websites, apps, TVs or other platforms with VAST or VMAP. Do not use for Dynamic Ad Insertion (DAI), SSAI, or … | [source](https://github.com/google/skills/tree/b15f327/skills/ads/interactive-media-ads/ima-sdk-basics) | + +## Source: veniceai/skills + +- Repository: [https://github.com/veniceai/skills](https://github.com/veniceai/skills) (commit `de089fa`, retrieved 2026-07-14) +- License: MIT + +| Skill | Tier | What it adds | Upstream | +|---|---|---|---| +| `venice-chat` | core | Call POST /chat/completions on Venice. Covers the OpenAI-compatible request shape, Venice-only venice_parameters (web search, E2EE, characters, thinking control, X search), multimodal inputs (images/audio/video), tool calls, reasoning … | [source](https://github.com/veniceai/skills/tree/de089fa/skills/venice-chat) | +| `venice-audio-music` | adjacent | Async music / audio-track generation via Venice. Covers the /audio/quote + /audio/queue + /audio/retrieve + /audio/complete lifecycle, lyrics vs instrumental, voice selection, duration, language, speed, model capability probing, and … | [source](https://github.com/veniceai/skills/tree/de089fa/skills/venice-audio-music) | +| `venice-audio-speech` | adjacent | Generate speech from text via POST /audio/speech. Covers TTS models (Kokoro, Qwen 3, xAI, Inworld, Chatterbox, Orpheus, ElevenLabs Turbo, MiniMax, Gemini Flash), voices per family, output formats (mp3/opus/aac/flac/wav/pcm), streaming, … | [source](https://github.com/veniceai/skills/tree/de089fa/skills/venice-audio-speech) | +| `venice-audio-transcription` | adjacent | Transcribe audio files to text via POST /audio/transcriptions. Covers supported models (Parakeet, Whisper, Wizper, Scribe, xAI STT), supported formats (wav/flac/m4a/aac/mp4/mp3/ogg/webm), response formats (json/text), timestamps, and … | [source](https://github.com/veniceai/skills/tree/de089fa/skills/venice-audio-transcription) | +| `venice-video` | adjacent | Generate and transcribe videos via Venice. Covers the async /video/quote + /video/queue + /video/retrieve + /video/complete loop, text-to-video, image-to-video, video-to-video (upscale), audio input, reference images, scene and element … | [source](https://github.com/veniceai/skills/tree/de089fa/skills/venice-video) | diff --git a/references/market.md b/references/market.md new file mode 100644 index 0000000..866ee20 --- /dev/null +++ b/references/market.md @@ -0,0 +1,62 @@ +# Market evidence report — boom-operator + +Source: **6 real job ads** (JSearch API, countries: us 6), extracted into the MSSQL evidence store; as of 2026-07-09. +This report contains extracted, aggregated facts only — no ad text is +reproduced (copyright / platform terms). + +## Seniority distribution + +| Seniority | Ads | Share | +|---|---|---| +| n/a | 5 | 83 % | +| mid | 1 | 17 % | + +## Tools — full market ranking + +| # | Item | Ads | Share | +|---|---|---|---| + +## Hard skills — full market ranking + +| # | Item | Ads | Share | +|---|---|---|---| +| 1 | boom operation | 3 | 50 % | + +## Methods — full market ranking + +| # | Item | Ads | Share | +|---|---|---|---| + +## Responsibilities — full market ranking + +| # | Item | Ads | Share | +|---|---|---|---| + +## Regional breakdown + +> **Corpus note:** 6 relevant ads in total — below the 100-ad target for a fully reliable ranking. Percentages above should be read as indicative. + +### US (us) + +**Insufficient evidence** — 6 ads (minimum for a regional ranking: 30). No ranking is reported for this region. + +### UK (gb) + +**Insufficient evidence** — 0 ads (minimum for a regional ranking: 30). No ranking is reported for this region. + +### EU/DACH (de, at, ch, nl) + +**Insufficient evidence** — 0 ads (minimum for a regional ranking: 30). No ranking is reported for this region. + + +## Job title variants in the market + +| Title | Ads | +|---|---| +| Sound Mixer / Boom Operator Job in One Last Time | 2 | +| Boom Operator Job in 'Call Me Back' | 1 | +| Boom operator Job in 'Escorting Yasilda | 1 | +| Boom Operator​/Utility | 1 | +| Sound Recordist / Boom Operator Job in Pardon | 1 | + +Methodology: entities extracted per ad ({hard_skills, tools, methods, responsibilities, seniority}), normalized, counted as DISTINCT ads per entity; report threshold ≥ 3 ads. Headline sections in skills.md/tools.md use the stricter ≥ 20 % threshold. diff --git a/references/profile.md b/references/profile.md new file mode 100644 index 0000000..24f4722 --- /dev/null +++ b/references/profile.md @@ -0,0 +1,23 @@ +# Occupation profile — boom operator + +- **ESCO URI:** http://data.europa.eu/esco/occupation/4d102082-8d43-4c59-81ab-08a7509f3c40 +- **ESCO code:** 3521.1.1 +- **ISCO-08 group:** 3521 — Broadcasting and audiovisual technicians + +## Description (ESCO) + +Boom operators set up and operate the boom microphone, either by hand, on an arm or on a moving platform. They make sure that every microphone is correctly stationed on set and in the best position to capture the dialogues. Boom operators are also responsible for the microphones on the actors clothing. + +## Definition + +nan + +## Alternative labels + +- TV boom operator +- shotgun microphone operator +- production sound mixer assistant +- shotgun operator +- boom holder +- sound equipment technician +- film boom operator diff --git a/references/skills.md b/references/skills.md new file mode 100644 index 0000000..2d43489 --- /dev/null +++ b/references/skills.md @@ -0,0 +1,70 @@ +# Competences — boom operator + +Source: ESCO v1.2.1 occupation-skill relations (http://data.europa.eu/esco/occupation/4d102082-8d43-4c59-81ab-08a7509f3c40). + +## Essential + +- **acoustics** (knowledge) +- **adapt to type of media** (skill/competence) +- **analyse a script** (skill/competence) +- **audiovisual equipment** (knowledge) +- **consult with sound editor** (skill/competence) +- **follow directions of the artistic director** (skill/competence) +- **follow work schedule** (skill/competence) +- **manage sound quality** (skill/competence) +- **perform soundchecks** (skill/competence) +- **program sound cues** (skill/competence) +- **set up sound equipment** (skill/competence) +- **study media sources** (skill/competence) +- **use audio reproduction software** (skill/competence) +- **use technical documentation** (skill/competence) +- **work ergonomically** (skill/competence) + +## Optional + +- adapt artistic plan to location (skill/competence) +- apply health and safety standards (skill/competence) +- assess power needs (skill/competence) +- assess sound quality (skill/competence) +- edit recorded sound (skill/competence) +- electricity (knowledge) +- health and safety regulations (knowledge) +- maintain sound equipment (skill/competence) +- operate an audio mixing console (skill/competence) +- supervise sound production (skill/competence) +- technically design a sound system (skill/competence) +- tune up wireless audio systems (skill/competence) + + + +## Market evidence (job-ad analysis, 6 ads, as of 2026-07-09) + +Share of analyzed job ads mentioning the item (threshold ≥ 20 %). Source: JSearch/Adzuna APIs. + +### Hard skills + +- boom operation — **50 %** +- dialogue capture — **33 %** +- sound mixing — **33 %** +- sound equipment operation — **17 %** +- boom pole operation — **17 %** +- microphone placement — **17 %** +- audio capture — **17 %** +- audio equipment operation — **17 %** +- audio quality control — **17 %** +- audio recording — **17 %** + +### Responsibilities + +- on-set sound recording — **33 %** +- audio recording — **17 %** +- boom microphone operation — **17 %** +- audio level maintenance — **17 %** +- boom operation — **17 %** +- microphone positioning — **17 %** +- noise reduction — **17 %** +- on-set audio handling — **17 %** +- actor movement tracking — **17 %** +- maintaining audio quality — **17 %** + + diff --git a/references/tasks.md b/references/tasks.md new file mode 100644 index 0000000..870156d --- /dev/null +++ b/references/tasks.md @@ -0,0 +1,32 @@ +# Tasks & work activities — boom operator + +Source: O*NET 30.3, occupation 27-4014.00 (Sound Engineering Technicians) — manual nearest-occupation mapping via ISCO group 3521; the official ESCO crosswalk has no entry for this ESCO occupation. + +## Task statements + +- **[Core]** Tear down equipment after event completion. +- **[Core]** Convert video and audio recordings into digital formats for editing or archiving. +- **[Core]** Set up, test, and adjust recording equipment for recording sessions and live performances. +- **[Core]** Confer with producers, performers, and others to determine and achieve the desired sound for a production, such as a musical recording or a film. +- **[Core]** Regulate volume level and sound quality during recording sessions, using control consoles. +- **[Core]** Prepare for recording sessions by performing such activities as selecting and setting up microphones. +- **[Core]** Report equipment problems and ensure that required repairs are made. +- **[Core]** Mix and edit voices, music, and taped sound effects for live performances and for prerecorded events, using sound mixing boards. +- **[Core]** Synchronize and equalize prerecorded dialogue, music, and sound effects with visual action of motion pictures or television productions, using control consoles. +- **[Core]** Record speech, music, and other sounds on recording media, using recording equipment. +- **[Core]** Reproduce and duplicate sound recordings from original recording media, using sound editing and duplication equipment. +- **[Core]** Separate instruments, vocals, and other sounds, and combine sounds during the mixing or postproduction stage. +- **[Core]** Keep logs of recordings. +- **[Supplemental]** Create musical instrument digital interface programs for music projects, commercials, or film postproduction. + +## Detailed work activities + +- Collaborate with others to determine technical details of productions. +- Convert data among multiple digital or analog formats. +- Dismantle equipment or temporary structures. +- Maintain logs of production activities. +- Mix sound inputs. +- Notify others of equipment problems. +- Operate audio recording equipment. +- Operate control consoles for sound, lighting or video. +- Select materials or props. diff --git a/references/tools.md b/references/tools.md new file mode 100644 index 0000000..be90e92 --- /dev/null +++ b/references/tools.md @@ -0,0 +1,38 @@ +# Tools & technology — boom operator + +Source: O*NET 30.3, occupation 27-4014.00 (Sound Engineering Technicians) — manual nearest-occupation mapping via ISCO group 3521; the official ESCO crosswalk has no entry for this ESCO occupation. + +| Software | Category | Hot technology | +|---|---|---| +| Adobe Acrobat | Document management software | yes | +| Adobe Creative Cloud software | Graphics or photo imaging software | yes | +| Adobe InDesign | Desktop publishing software | yes | +| Adobe Photoshop | Graphics or photo imaging software | yes | +| Apple macOS | Operating system software | yes | +| Autodesk AutoCAD | Computer aided design CAD software | yes | +| Facebook | Web page creation and editing software | yes | +| Git | File versioning software | yes | +| Linux | Operating system software | yes | +| Microsoft Excel | Spreadsheet software | yes | +| Microsoft Office software | Office suite software | yes | +| Microsoft PowerPoint | Presentation software | yes | +| Microsoft Visio | Process mapping and design software | yes | +| Microsoft Windows | Operating system software | yes | +| Microsoft Word | Word processing software | yes | +| Oracle Java | Object or component oriented development software | yes | +| UNIX | Operating system software | yes | +| Adobe Audition | Music or sound editing software | | +| Adobe Premiere Pro | Video creation and editing software | | +| Apple Final Cut Pro | Video creation and editing software | | +| Audio editing software | Music or sound editing software | | +| Avid Pro Tools | Music or sound editing software | | +| Avid Technology audio visual editing software | Video creation and editing software | | +| Avid Technology Pro Tools | Music or sound editing software | | +| Cisco IOS | Operating system software | | +| IBM Middleware | Transaction server software | | +| Musical instrument digital interface MIDI software | Music or sound editing software | | +| Perforce software | Metadata management software | | +| Real time operating system RTOS software | Operating system software | | +| Unity Technologies Unity | Development environment software | | +| VMware | Clustering software | | +| Voice over internet protocol VoIP system software | Internet protocol IP multimedia subsystem software | |