From 7dd12018490a02894f3bb929f789ff8bd0cbd325 Mon Sep 17 00:00:00 2001 From: Bill Peck Date: Fri, 9 Oct 2026 09:23:46 -0400 Subject: [PATCH 1/3] feat(azure_rm_searchskillset): add skillset and info modules Add azure_rm_searchskillset and azure_rm_searchskillset_info to manage enrichment skillsets in an Azure AI Search service (data plane), built on the azure-search-documents SDK via the shared AzureRMSearchDataPlaneMixin (SearchIndexerClient), mirroring the merged azure_rm_searchindex pattern. Idempotent and check-mode aware. Azure redacts cognitiveServices.key on read, so it is dropped from the drift comparison; on a full-replace PUT update the module merges user-supplied keys onto the existing skillset to preserve unsupplied settings (skills, description, knowledgeStore). Because there is no "keep" sentinel for the key, the existing cognitiveServices is never echoed back; it is re-sent only when the user supplies a key. skills is required only on create. Co-Authored-By: Claude Opus 4.8 --- meta/runtime.yml | 2 + plugins/modules/azure_rm_searchskillset.py | 272 ++++++++++++++++++ .../modules/azure_rm_searchskillset_info.py | 123 ++++++++ 3 files changed, 397 insertions(+) create mode 100644 plugins/modules/azure_rm_searchskillset.py create mode 100644 plugins/modules/azure_rm_searchskillset_info.py diff --git a/meta/runtime.yml b/meta/runtime.yml index bbe1900e8..e6e02d7fe 100644 --- a/meta/runtime.yml +++ b/meta/runtime.yml @@ -343,6 +343,8 @@ action_groups: - azure.azcollection.azure_rm_routetable_info - azure.azcollection.azure_rm_searchindex - azure.azcollection.azure_rm_searchindex_info + - azure.azcollection.azure_rm_searchskillset + - azure.azcollection.azure_rm_searchskillset_info - azure.azcollection.azure_rm_securitygroup - azure.azcollection.azure_rm_securitygroup_info - azure.azcollection.azure_rm_servicebus diff --git a/plugins/modules/azure_rm_searchskillset.py b/plugins/modules/azure_rm_searchskillset.py new file mode 100644 index 000000000..6c2926f1c --- /dev/null +++ b/plugins/modules/azure_rm_searchskillset.py @@ -0,0 +1,272 @@ +#!/usr/bin/python +# +# Copyright (c) 2026 Bill Peck (@p3ck) +# +# GNU General Public License v3.0+ (see COPYING or https://www.gnu.org/licenses/gpl-3.0.txt) + +from __future__ import absolute_import, division, print_function +__metaclass__ = type + +DOCUMENTATION = ''' +--- +module: azure_rm_searchskillset +version_added: "4.2.0" +short_description: Manage a skillset in an Azure AI Search service +description: + - Create, update, and delete a skillset within an Azure AI Search service + (data plane). A skillset is an ordered set of enrichment skills (OCR, + entity recognition, text split, embeddings, custom Web API skills, ...) + that an indexer applies to documents during ingestion. +options: + resource_group: + description: + - Name of the resource group containing the search service. + required: true + type: str + search_service_name: + description: + - Name of the Azure AI Search service that hosts the skillset. + required: true + type: str + name: + description: + - Name of the skillset. + required: true + type: str + admin_key: + description: + - Admin API key for the search service, used to authenticate data-plane + requests via the C(api-key) header. + - This is supplementary to the standard Azure credentials. The module + always requires standard Azure authentication parameters and a + subscription ID (see the I(azure.azcollection.azure) documentation + fragment) to run, regardless of whether this is set. + - If omitted, data-plane requests are authenticated with an RBAC bearer + token (managed identity / service principal) using the data-plane + scope C(https://search.azure.com/.default). + type: str + skills: + description: + - The ordered list of skills in the skillset. Required when creating a skillset. + - Each skill is a dict passed through to the Azure AI Search API verbatim, + keyed by its C(@odata.type) (for example + C(#Microsoft.Skills.Text.SplitSkill)). See the Azure AI Search + reference for each skill's inputs, outputs, and parameters. + type: list + elements: dict + description: + description: + - Free-text description of the skillset. + type: str + cognitive_services: + description: + - Reference to an Azure AI (Cognitive) Services account used to bill + the billable skills in the skillset. + - Sent to Azure as C(cognitiveServices) using the by-key resource + reference. Azure redacts the key on read, so changes to the key + alone are not detected as drift. + - On update, this must be resupplied to retain a key-based binding; + if omitted, it is not re-sent and the skillset reverts to the + default (free) cognitive services allocation. + type: dict + suboptions: + key: + description: API key of the Azure AI (Cognitive) Services account. + type: str + required: true + knowledge_store: + description: + - Knowledge store definition (projections to Azure Storage). + - Passed through to the Azure AI Search API using camelCase keys; see the Azure AI Search reference for structure. + type: dict + state: + description: + - Assert the state of the skillset. Use C(present) to create/update, C(absent) to delete. + type: str + default: present + choices: + - present + - absent +extends_documentation_fragment: + - azure.azcollection.azure +author: + - Bill Peck (@p3ck) +''' + +EXAMPLES = ''' +- name: Create a text-split skillset + azure.azcollection.azure_rm_searchskillset: + resource_group: myResourceGroup + search_service_name: mysearchsvc + name: chunking-skillset + skills: + - "@odata.type": "#Microsoft.Skills.Text.SplitSkill" + context: /document + textSplitMode: pages + maximumPageLength: 1000 + inputs: + - name: text + source: /document/content + outputs: + - name: textItems + targetName: pages + state: present + +- name: Delete a skillset + azure.azcollection.azure_rm_searchskillset: + resource_group: myResourceGroup + search_service_name: mysearchsvc + name: chunking-skillset + state: absent +''' + +RETURN = ''' +state: + description: + - The skillset definition as returned by Azure AI Search. + - The cognitive services key is redacted by Azure and is not returned in clear text. + returned: when I(state=present) + type: dict + sample: {"name": "chunking-skillset", "skills": [{"@odata.type": "#Microsoft.Skills.Text.SplitSkill"}]} +''' + +import copy + +from ansible_collections.azure.azcollection.plugins.module_utils.azure_rm_common_ext import AzureRMModuleBaseExt +from ansible_collections.azure.azcollection.plugins.module_utils.azure_rm_search_common import ( + AzureRMSearchDataPlaneMixin, + ResourceNotFoundError, +) + + +class AzureRMSearchSkillset(AzureRMSearchDataPlaneMixin, AzureRMModuleBaseExt): + + def __init__(self): + self.module_arg_spec = dict( + resource_group=dict(type='str', required=True), + search_service_name=dict(type='str', required=True), + name=dict(type='str', required=True), + admin_key=dict(type='str', no_log=True), + skills=dict(type='list', elements='dict'), + description=dict(type='str'), + cognitive_services=dict(type='dict', options=dict( + key=dict(type='str', required=True, no_log=True), + )), + knowledge_store=dict(type='dict'), + state=dict(type='str', default='present', choices=['present', 'absent']), + ) + self.resource_group = None + self.search_service_name = None + self.name = None + self.admin_key = None + self.skills = None + self.description = None + self.cognitive_services = None + self.knowledge_store = None + self.state = None + self.results = dict(changed=False) + super(AzureRMSearchSkillset, self).__init__( + derived_arg_spec=self.module_arg_spec, + supports_check_mode=True, + supports_tags=False, + ) + + def exec_module(self, **kwargs): + for key in list(self.module_arg_spec.keys()): + setattr(self, key, kwargs[key]) + + self.client = self.get_search_indexer_client( + self.search_service_name, admin_key=self.admin_key) + + existing = self._get_existing() + + if self.state == 'present': + if self.skills is None and existing is None: + self.fail(msg="skills is required to create skillset '%s'; " + "it does not exist yet" % self.name) + desired = self._build_body() + if existing is None: + self.results['changed'] = True + if not self.check_mode: + self.results['state'] = self._create_or_update(desired) + else: + self.results['state'] = desired + else: + if not self._is_current(desired, existing): + # create_or_update is a full-replace PUT; merge the user- + # supplied keys onto the existing skillset so unsupplied + # settings (skills, description, knowledgeStore) are preserved. + body = self._merge_existing(desired, existing) + self.results['changed'] = True + if not self.check_mode: + self.results['state'] = self._create_or_update(body) + else: + self.results['state'] = body + else: + self.results['state'] = existing + else: # absent + if existing is not None: + self.results['changed'] = True + if not self.check_mode: + self.client.delete_skillset(self.name) + return self.results + + def _get_existing(self): + # The SDK raises ResourceNotFoundError when the skillset does not exist; + # translate to None. Otherwise return the camelCase wire shape. + try: + return self.client.get_skillset(self.name).as_dict() + except ResourceNotFoundError: + return None + + def _build_body(self): + body = {"name": self.name} + if self.skills is not None: + body["skills"] = self.skills + if self.description is not None: + body["description"] = self.description + if self.cognitive_services is not None and self.cognitive_services.get("key") is not None: + body["cognitiveServices"] = { + "@odata.type": "#Microsoft.Azure.Search.CognitiveServicesByKey", + "key": self.cognitive_services["key"], + } + if self.knowledge_store is not None: + body["knowledgeStore"] = self.knowledge_store + return body + + def _merge_existing(self, desired, existing): + # Overlay the user-supplied keys onto a copy of the existing skillset so + # a PUT update does not drop settings the user did not resupply (skills, + # description, knowledgeStore). Strip response-only @odata.* annotations. + # Azure redacts cognitiveServices.key on read and there is no "keep" + # sentinel, so never echo the existing cognitiveServices; it is re-sent + # only when the user supplies a key in `desired`. + merged = {k: v for k, v in copy.deepcopy(existing).items() + if not k.startswith('@odata.')} + merged.pop('cognitiveServices', None) + merged.update(desired) + return merged + + def _create_or_update(self, body): + # create_or_update_skillset accepts a plain camelCase dict and returns + # the full model; .as_dict() yields the wire shape. + return self.client.create_or_update_skillset(body).as_dict() + + def _is_current(self, desired, existing): + # Azure redacts cognitiveServices.key on read, so it can never match + # what we would PUT. Drop cognitiveServices from the comparison; every + # other field is compared via default_compare (union-of-keys walk, so + # server defaults present only in `existing` are ignored). Deep-copy so + # the comparison cannot mutate the body we would PUT. + compare_body = copy.deepcopy(desired) + compare_body.pop("cognitiveServices", None) + result = dict(compare=[]) + return self.default_compare({}, compare_body, existing, '', result) + + +def main(): + AzureRMSearchSkillset() + + +if __name__ == '__main__': + main() diff --git a/plugins/modules/azure_rm_searchskillset_info.py b/plugins/modules/azure_rm_searchskillset_info.py new file mode 100644 index 000000000..821d830df --- /dev/null +++ b/plugins/modules/azure_rm_searchskillset_info.py @@ -0,0 +1,123 @@ +#!/usr/bin/python +# +# Copyright (c) 2026 Bill Peck (@p3ck) +# +# GNU General Public License v3.0+ (see COPYING or https://www.gnu.org/licenses/gpl-3.0.txt) + +from __future__ import absolute_import, division, print_function +__metaclass__ = type + +DOCUMENTATION = ''' +--- +module: azure_rm_searchskillset_info +version_added: "4.2.0" +short_description: Get information about Azure AI Search skillsets +description: + - Get details of a single skillset, or list all skillsets in an Azure AI Search service. +options: + resource_group: + description: + - Name of the resource group containing the search service. + required: true + type: str + search_service_name: + description: + - Name of the Azure AI Search service. + required: true + type: str + name: + description: + - Name of a specific skillset to fetch. If omitted, all skillsets are listed. + type: str + admin_key: + description: + - Admin API key for the search service, used to authenticate data-plane + requests via the C(api-key) header. + - This is supplementary to the standard Azure credentials. The module + always requires standard Azure authentication parameters and a + subscription ID (see the I(azure.azcollection.azure) documentation + fragment) to run, regardless of whether this is set. + - If omitted, data-plane requests are authenticated with an RBAC bearer + token (managed identity / service principal) using the data-plane + scope C(https://search.azure.com/.default). + type: str +extends_documentation_fragment: + - azure.azcollection.azure +author: + - Bill Peck (@p3ck) +''' + +EXAMPLES = ''' +- name: List all skillsets + azure.azcollection.azure_rm_searchskillset_info: + resource_group: myResourceGroup + search_service_name: mysearchsvc + +- name: Get one skillset + azure.azcollection.azure_rm_searchskillset_info: + resource_group: myResourceGroup + search_service_name: mysearchsvc + name: chunking-skillset +''' + +RETURN = ''' +skillsets: + description: + - List of skillset definitions. + - The cognitive services key is redacted by Azure and is not returned in clear text. + returned: always + type: list + elements: dict + sample: [{"name": "chunking-skillset", "skills": [{"@odata.type": "#Microsoft.Skills.Text.SplitSkill"}]}] +''' + +from ansible_collections.azure.azcollection.plugins.module_utils.azure_rm_common_ext import AzureRMModuleBaseExt +from ansible_collections.azure.azcollection.plugins.module_utils.azure_rm_search_common import ( + AzureRMSearchDataPlaneMixin, + ResourceNotFoundError, +) + + +class AzureRMSearchSkillsetInfo(AzureRMSearchDataPlaneMixin, AzureRMModuleBaseExt): + + def __init__(self): + self.module_arg_spec = dict( + resource_group=dict(type='str', required=True), + search_service_name=dict(type='str', required=True), + name=dict(type='str'), + admin_key=dict(type='str', no_log=True), + ) + self.resource_group = None + self.search_service_name = None + self.name = None + self.admin_key = None + self.results = dict(changed=False, skillsets=[]) + super(AzureRMSearchSkillsetInfo, self).__init__( + derived_arg_spec=self.module_arg_spec, + supports_check_mode=True, + supports_tags=False, + facts_module=True, + ) + + def exec_module(self, **kwargs): + for key in list(self.module_arg_spec.keys()): + setattr(self, key, kwargs[key]) + client = self.get_search_indexer_client( + self.search_service_name, admin_key=self.admin_key) + if self.name: + try: + ss = client.get_skillset(self.name) + self.results['skillsets'] = [ss.as_dict()] + except ResourceNotFoundError: + self.results['skillsets'] = [] + else: + self.results['skillsets'] = [ss.as_dict() for ss in client.get_skillsets()] + return self.results + + +def main(): + AzureRMSearchSkillsetInfo() + + +if __name__ == '__main__': + main() From a0d3285b5e4d2920cd82565e880b5170a182599f Mon Sep 17 00:00:00 2001 From: Bill Peck Date: Fri, 9 Oct 2026 09:23:47 -0400 Subject: [PATCH 2/3] test(azure_rm_searchskillset): add integration target Stands up a search service (AAD data-plane auth + Search Service Contributor) and exercises create/idempotent/info/update/delete with a self-contained text-split skill. The update changes only the description (omitting skills) to verify the PUT merge preserves the skills. Co-Authored-By: Claude Opus 4.8 --- .../targets/azure_rm_searchskillset/aliases | 3 + .../azure_rm_searchskillset/meta/main.yml | 2 + .../azure_rm_searchskillset/tasks/main.yml | 225 ++++++++++++++++++ 3 files changed, 230 insertions(+) create mode 100644 tests/integration/targets/azure_rm_searchskillset/aliases create mode 100644 tests/integration/targets/azure_rm_searchskillset/meta/main.yml create mode 100644 tests/integration/targets/azure_rm_searchskillset/tasks/main.yml diff --git a/tests/integration/targets/azure_rm_searchskillset/aliases b/tests/integration/targets/azure_rm_searchskillset/aliases new file mode 100644 index 000000000..f4b2242f5 --- /dev/null +++ b/tests/integration/targets/azure_rm_searchskillset/aliases @@ -0,0 +1,3 @@ +cloud/azure +shippable/azure/group17 +destructive diff --git a/tests/integration/targets/azure_rm_searchskillset/meta/main.yml b/tests/integration/targets/azure_rm_searchskillset/meta/main.yml new file mode 100644 index 000000000..95e1952f9 --- /dev/null +++ b/tests/integration/targets/azure_rm_searchskillset/meta/main.yml @@ -0,0 +1,2 @@ +dependencies: + - setup_azure diff --git a/tests/integration/targets/azure_rm_searchskillset/tasks/main.yml b/tests/integration/targets/azure_rm_searchskillset/tasks/main.yml new file mode 100644 index 000000000..e0a74cea4 --- /dev/null +++ b/tests/integration/targets/azure_rm_searchskillset/tasks/main.yml @@ -0,0 +1,225 @@ +- name: Create a resource group for the search skillset test + azure.azcollection.azure_rm_resourcegroup: + name: "{{ resource_group }}-searchss" + location: eastus + +- name: Run search skillset tests + block: + - name: Build a unique search service name + ansible.builtin.set_fact: + svc_name: "searchss{{ resource_group | hash('md5') | truncate(14, True, '') }}" + + - name: Create AI Search service + azure.azcollection.azure_rm_cognitivesearch: + resource_group: "{{ resource_group }}-searchss" + name: "{{ svc_name }}" + sku: basic + register: svc + + # Data-plane RBAC is off by default on a new search service (API-key only), + # so enable AAD auth and grant the test principal a data-plane role. This + # is what exercises the https://search.azure.com/.default token path. + - name: Enable AAD (RBAC) data-plane auth on the search service + azure.azcollection.azure_rm_resource: + api_version: '2023-11-01' + resource_group: "{{ resource_group }}-searchss" + provider: search + resource_type: searchservices + resource_name: "{{ svc_name }}" + method: PATCH + body: + properties: + authOptions: + aadOrApiKey: + aadAuthFailureMode: http403 + + - name: Resolve the test service principal object id + azure.azcollection.azure_rm_adserviceprincipal_info: + app_id: "{{ azure_client_id }}" + register: sp_info + + - name: Grant Search Service Contributor to the test principal + azure.azcollection.azure_rm_roleassignment: + scope: "{{ svc.state.id }}" + assignee_object_id: "{{ sp_info.service_principals[0].object_id }}" + # Search Service Contributor: manage data-plane object definitions + # (indexes, data sources, indexers, skillsets). + # https://learn.microsoft.com/en-us/azure/search/search-security-rbac + role_definition_id: "/subscriptions/{{ azure_subscription_id }}/providers/Microsoft.Authorization/roleDefinitions/7ca78c08-252a-4471-8644-bb5ff32d4ba0" + state: present + + # A text-split skill needs no Azure AI Services account, so the skillset is + # self-contained (no secrets, clean idempotency). + - name: Create skillset (check mode) + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + skills: + - "@odata.type": "#Microsoft.Skills.Text.SplitSkill" + context: /document + textSplitMode: pages + maximumPageLength: 1000 + inputs: + - name: text + source: /document/content + outputs: + - name: textItems + targetName: pages + check_mode: true + register: ss_check + # Absorb AAD auth-option and role-assignment propagation (can take minutes). + retries: 20 + delay: 15 + until: ss_check is not failed + + - name: Assert check mode reports change without creating + ansible.builtin.assert: + that: + - ss_check.changed + + - name: Create skillset + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + skills: + - "@odata.type": "#Microsoft.Skills.Text.SplitSkill" + context: /document + textSplitMode: pages + maximumPageLength: 1000 + inputs: + - name: text + source: /document/content + outputs: + - name: textItems + targetName: pages + register: ss_create + + - name: Assert created (RBAC data-plane scope worked) + ansible.builtin.assert: + that: + - ss_create.changed + - ss_create.state.name == 'it-skillset' + - ss_create.state.skills | length == 1 + + - name: Create skillset again (idempotent) + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + skills: + - "@odata.type": "#Microsoft.Skills.Text.SplitSkill" + context: /document + textSplitMode: pages + maximumPageLength: 1000 + inputs: + - name: text + source: /document/content + outputs: + - name: textItems + targetName: pages + register: ss_idem + + - name: Assert no change on second run + ansible.builtin.assert: + that: + - not ss_idem.changed + + - name: Get skillset via info + azure.azcollection.azure_rm_searchskillset_info: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + register: ss_info + + - name: Assert info returns the skillset + ansible.builtin.assert: + that: + - ss_info.skillsets | length == 1 + - ss_info.skillsets[0].name == 'it-skillset' + + # Change ONLY the description, omitting skills. create_or_update is a full- + # replace PUT, so this verifies the module carries forward the existing + # skills instead of dropping them. + - name: Update the skillset (description only) + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + description: Document chunking skillset + register: ss_update + + - name: Assert update changed and preserved skills + ansible.builtin.assert: + that: + - ss_update.changed + - ss_update.state.description == 'Document chunking skillset' + - ss_update.state.skills | length == 1 + + - name: Update the skillset again (idempotent) + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + description: Document chunking skillset + register: ss_update_idem + + - name: Assert no change on repeated description-only update + ansible.builtin.assert: + that: + - not ss_update_idem.changed + + - name: Info for a non-existent skillset returns empty + azure.azcollection.azure_rm_searchskillset_info: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: does-not-exist + register: ss_missing + + - name: Assert empty list, not failure + ansible.builtin.assert: + that: + - ss_missing.skillsets | length == 0 + + - name: Delete skillset + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + state: absent + register: ss_delete + + - name: Assert deleted + ansible.builtin.assert: + that: + - ss_delete.changed + + - name: Delete skillset again (idempotent absent) + azure.azcollection.azure_rm_searchskillset: + resource_group: "{{ resource_group }}-searchss" + search_service_name: "{{ svc_name }}" + name: it-skillset + state: absent + register: ss_delete_idem + + - name: Assert no change on second delete + ansible.builtin.assert: + that: + - not ss_delete_idem.changed + + always: + - name: Clean up search service + azure.azcollection.azure_rm_cognitivesearch: + resource_group: "{{ resource_group }}-searchss" + name: "{{ svc_name }}" + state: absent + ignore_errors: true + + - name: Clean up resource group + azure.azcollection.azure_rm_resourcegroup: + name: "{{ resource_group }}-searchss" + location: eastus + state: absent + force_delete_nonempty: true + ignore_errors: true From f1dbc7a68aa0a0bce4ee9f6aa1828b7619e40fc4 Mon Sep 17 00:00:00 2001 From: Bill Peck Date: Fri, 9 Oct 2026 09:23:47 -0400 Subject: [PATCH 3/3] ci(azure_rm_searchskillset): add integration target to pr-pipelines Co-Authored-By: Claude Opus 4.8 --- pr-pipelines.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/pr-pipelines.yml b/pr-pipelines.yml index 1c70d4079..421f48a7c 100644 --- a/pr-pipelines.yml +++ b/pr-pipelines.yml @@ -206,6 +206,7 @@ parameters: - "azure_rm_tags" - "azure_rm_dedicatedhost" - "azure_rm_searchindex" + - "azure_rm_searchskillset" - "inventory_azure" - "setup_azure"