diff --git a/docs_user/assemblies/assembly_adopting-key-manager-service-with-hsm.adoc b/docs_user/assemblies/assembly_adopting-key-manager-service-with-hsm.adoc index a4ed1e68f..41a147a0e 100644 --- a/docs_user/assemblies/assembly_adopting-key-manager-service-with-hsm.adoc +++ b/docs_user/assemblies/assembly_adopting-key-manager-service-with-hsm.adoc @@ -10,22 +10,25 @@ ifdef::context[:parent-context: {context}] [role="_abstract"] Adopt the {key_manager_first_ref} from {OpenStackPreviousInstaller} to {rhos_long} when your source environment includes hardware security module (HSM) integration to preserve HSM functionality and maintain access to HSM-backed secrets. HSM provides enhanced security for cryptographic operations by storing encryption keys in dedicated hardware devices. +For additional information about the {key_manager} before you start the adoption, see the following resources: + +* {key_manager} service configuration documentation +* Hardware security module vendor-specific documentation +* OpenStack Barbican PKCS#11 plugin documentation + include::../modules/con_key-manager-service-hsm-adoption-approaches.adoc[leveloffset=+1] include::../modules/proc_adopting-key-manager-service-with-proteccio-hsm.adoc[leveloffset=+1] include::../modules/proc_adopting-key-manager-service-with-hsm-integration.adoc[leveloffset=+1] -include::../modules/ref_troubleshooting-key-manager-hsm-adoption.adoc[leveloffset=+1] +include::../assemblies/assembly_troubleshooting-key-manager-hsm-adoption.adoc[leveloffset=+1] -include::../modules/ref_troubleshooting-key-manager-proteccio-adoption.adoc[leveloffset=+1] +include::../assemblies/assembly_troubleshooting-key-manager-proteccio-adoption.adoc[leveloffset=+1] + +include::../modules/proc_rolling-back-the-hsm-adoption.adoc[leveloffset=+1] -[role="_additional-resources"] -== Additional resources -* {key_manager} service configuration documentation -* Hardware security module vendor-specific documentation -* OpenStack Barbican PKCS#11 plugin documentation ifdef::parent-context[:context: {parent-context}] ifndef::parent-context[:!context:] diff --git a/docs_user/assemblies/assembly_adopting-openstack-control-plane-services.adoc b/docs_user/assemblies/assembly_adopting-openstack-control-plane-services.adoc index 207cb382b..e17d6a308 100644 --- a/docs_user/assemblies/assembly_adopting-openstack-control-plane-services.adoc +++ b/docs_user/assemblies/assembly_adopting-openstack-control-plane-services.adoc @@ -20,6 +20,8 @@ include::../assemblies/assembly_adopting-key-manager-service-with-hsm.adoc[level include::../modules/proc_adopting-the-networking-service.adoc[leveloffset=+1] +include::../modules/proc_configuring-control-plane-networking-for-spine-leaf.adoc[leveloffset=+1] + include::../modules/proc_adopting-the-object-storage-service.adoc[leveloffset=+1] include::../assemblies/assembly_adopting-the-image-service.adoc[leveloffset=+1] diff --git a/docs_user/assemblies/assembly_adopting-the-data-plane.adoc b/docs_user/assemblies/assembly_adopting-the-data-plane.adoc index 2212a3510..2fd72e8c0 100644 --- a/docs_user/assemblies/assembly_adopting-the-data-plane.adoc +++ b/docs_user/assemblies/assembly_adopting-the-data-plane.adoc @@ -24,11 +24,15 @@ include::../modules/proc_stopping-infrastructure-management-and-compute-services include::../modules/proc_adopting-compute-services-to-the-data-plane.adoc[leveloffset=+1] +include::../modules/proc_configuring-dcn-data-plane-nodesets.adoc[leveloffset=+1] + include::../modules/proc_performing-a-fast-forward-upgrade-on-compute-services.adoc[leveloffset=+1] include::../modules/proc_adopting-networker-services-to-the-data-plane.adoc[leveloffset=+1] include::../modules/proc_enabling-high-availability-for-instances.adoc[leveloffset=+1] +include::../modules/proc_performing-post-adoption-cleanup-of-load-balancers.adoc[leveloffset=+1] + ifdef::parent-context[:context: {parent-context}] ifndef::parent-context[:!context:] diff --git a/docs_user/assemblies/assembly_rhoso-180-adoption-overview.adoc b/docs_user/assemblies/assembly_rhoso-180-adoption-overview.adoc index 9711998e3..5fe5811c9 100644 --- a/docs_user/assemblies/assembly_rhoso-180-adoption-overview.adoc +++ b/docs_user/assemblies/assembly_rhoso-180-adoption-overview.adoc @@ -24,12 +24,16 @@ include::../modules/con_adoption-guidelines.adoc[leveloffset=+1] include::../modules/con_adoption-process-overview.adoc[leveloffset=+1] +include::../modules/con_dcn-adoption-overview.adoc[leveloffset=+1] + include::../modules/proc_installing-the-systemd-container-package-on-compute-hosts.adoc[leveloffset=+1] include::../modules/con_identity-service-authentication.adoc[leveloffset=+1] include::../assemblies/assembly_configuring-network-for-RHOSO-deployment.adoc[leveloffset=+1] +include::../modules/con_adopting-spine-leaf-networks.adoc[leveloffset=+1] + include::../assemblies/assembly_storage-requirements.adoc[leveloffset=+1] include::../assemblies/assembly_red-hat-ceph-storage-prerequisites.adoc[leveloffset=+1] @@ -38,5 +42,7 @@ include::../assemblies/assembly_preparing-an-instance-HA-deployment-for-adoption include::../modules/proc_comparing-configuration-files-between-deployments.adoc[leveloffset=+1] +include::../modules/con_preventing-config-loss-when-using-oc-patch.adoc[leveloffset=+1] + ifdef::parent-context[:context: {parent-context}] ifndef::parent-context[:!context:] diff --git a/docs_user/assemblies/assembly_troubleshooting-key-manager-hsm-adoption.adoc b/docs_user/assemblies/assembly_troubleshooting-key-manager-hsm-adoption.adoc new file mode 100644 index 000000000..244f93f1d --- /dev/null +++ b/docs_user/assemblies/assembly_troubleshooting-key-manager-hsm-adoption.adoc @@ -0,0 +1,41 @@ +:_mod-docs-content-type: ASSEMBLY +ifdef::context[:parent-context: {context}] + +[id="troubleshooting-key-manager-hsm-adoption_{context}"] + += Troubleshooting Key Manager HSM adoption + +:context: troubleshooting-hsm + +[role="_abstract"] +Review troubleshooting guidance for common issues that you might encounter while you perform the HSM-enabled Key Manager (Barbican) service adoption. + +If issues persist after following the troubleshooting guide: + +* Collect adoption logs and configuration for analysis. +* Check the HSM vendor documentation for vendor-specific troubleshooting. +* Verify HSM server status and connectivity independently. +* Review the adoption summary report for additional diagnostic information. + +include::../modules/proc_resolving-config-validation-failures.adoc[leveloffset=+1] + +include::../modules/proc_resolving-missing-HSM-file-prerequisites.adoc[leveloffset=+1] + +include::../modules/proc_resolving-connectivity-issues.adoc[leveloffset=+1] + +include::../modules/proc_resolving-HSM-secret-creation-failures.adoc[leveloffset=+1] + +include::../modules/proc_resolving-custom-image-registry-issues.adoc[leveloffset=+1] + +include::../modules/proc_resolving-hsm-backend-detection-failures.adoc[leveloffset=+1] + +include::../modules/proc_resolving-database-migration-issues.adoc[leveloffset=+1] + +include::../modules/proc_resolving-service-startup-failures.adoc[leveloffset=+1] + +include::../modules/proc_resolving-performance-and-connectivity-issues.adoc[leveloffset=+1] + + + +ifdef::parent-context[:context: {parent-context}] +ifndef::parent-context[:!context:] diff --git a/docs_user/assemblies/assembly_troubleshooting-key-manager-proteccio-adoption.adoc b/docs_user/assemblies/assembly_troubleshooting-key-manager-proteccio-adoption.adoc new file mode 100644 index 000000000..4d589bc89 --- /dev/null +++ b/docs_user/assemblies/assembly_troubleshooting-key-manager-proteccio-adoption.adoc @@ -0,0 +1,23 @@ +:_mod-docs-content-type: ASSEMBLY +ifdef::context[:parent-context: {context}] + +[id="troubleshooting-key-manager-proteccio-adoption_{context}"] + +:context: troubleshooting-proteccio + += Troubleshooting {key_manager} Proteccio HSM adoption + +[role="_abstract"] +Use this reference to troubleshoot common issues that might occur during {key_manager_first_ref} adoption with Proteccio HSM integration. If Proteccio HSM issues persist, consult the Eviden Trustway documentation and ensure that HSM server configuration matches the client settings. + +include::../modules/proc_resolving-prerequisite-validation-failures.adoc[leveloffset=+1] +include::../modules/proc_resolving-ssh-connection-failures.adoc[leveloffset=+1] +include::../modules/proc_resolving-database-import-failures.adoc[leveloffset=+1] +include::../modules/proc_resolving-custom-image-pull-failures.adoc[leveloffset=+1] +include::../modules/proc_resolving-hsm-certificate-mounting-issues.adoc[leveloffset=+1] +include::../modules/proc_resolving-service-startup-failures.adoc[leveloffset=+1] +include::../modules/proc_resolving-adoption-verification-failures.adoc[leveloffset=+1] + + +ifdef::parent-context[:context: {parent-context}] +ifndef::parent-context[:!context:] diff --git a/docs_user/modules/con_adopting-spine-leaf-networks.adoc b/docs_user/modules/con_adopting-spine-leaf-networks.adoc new file mode 100644 index 000000000..513ee78f1 --- /dev/null +++ b/docs_user/modules/con_adopting-spine-leaf-networks.adoc @@ -0,0 +1,123 @@ +:_mod-docs-content-type: CONCEPT +[id="adopting-spine-leaf-networks_{context}"] + += Configuring spine-leaf networks for the {rhos_long_noacro} deployment + +[role="_abstract"] +When you adopt a {rhos_prev_long} ({OpenStackShort}) deployment with spine-leaf networking, like a Distributed Compute Node (DCN) architecture, you must each L2 network segment with a separate IP subnet and create create routed provider networks. Traffic between sites is routed at L3 through spine routers or similar network infrastructure. + +You must configure routing for Compute nodes at edge sites to connect with control plane services, such as RabbitMQ or the database at the central site. The cloud will not function correctly without routes configured. + +[NOTE] +==== +DHCP relay is not supported in adopted {rhos_long} environments with spine-leaf topologies. This affects bare-metal provisioning scenarios that use PXE boot. + +If you need to provision bare-metal nodes at edge sites, use Redfish virtual media or similar BMC virtual media features instead of PXE boot. +==== + +.Example routes required on DCN1 Compute nodes +[options="header"] +|=== +| Destination network | Next hop | Purpose +| 172.17.0.0/24 | 172.17.10.1 | Route to central internalapi +| 172.17.20.0/24 | 172.17.10.1 | Route to DCN2 internalapi +| 172.18.0.0/24 | 172.18.10.1 | Route to central storage +| 172.18.20.0/24 | 172.18.10.1 | Route to DCN2 storage +|=== + +You configure these routes in the `edpm_network_config_template` within the `OpenStackDataPlaneNodeSet` custom resource (CR) for each site. + +.Example network topology for a three-site DCN deployment +[options="header"] +|=== +| Network | Central site | DCN1 site | DCN2 site +| Control plane | 192.168.122.0/24 | 192.168.133.0/24 | 192.168.144.0/24 +| Internal API | 172.17.0.0/24 | 172.17.10.0/24 | 172.17.20.0/24 +| Storage | 172.18.0.0/24 | 172.18.10.0/24 | 172.18.20.0/24 +| Tenant | 172.19.0.0/24 | 172.19.10.0/24 | 172.19.20.0/24 +|=== + +When you adopt a spine-leaf deployment, you configure the `NetConfig` CR with multiple subnets for each service network. Each subnet represents a different site. + +.Example NetConfig with multiple subnets per network +[source,yaml] +---- +apiVersion: network.openstack.org/v1beta1 +kind: NetConfig +metadata: + name: netconfig +spec: + networks: + - name: ctlplane + dnsDomain: ctlplane.example.com + subnets: + - name: subnet1 # Central site + allocationRanges: + - end: 192.168.122.120 + start: 192.168.122.100 + cidr: 192.168.122.0/24 + gateway: 192.168.122.1 + - name: ctlplanedcn1 # DCN1 site + allocationRanges: + - end: 192.168.133.120 + start: 192.168.133.100 + cidr: 192.168.133.0/24 + gateway: 192.168.133.1 + - name: ctlplanedcn2 # DCN2 site + allocationRanges: + - end: 192.168.144.120 + start: 192.168.144.100 + cidr: 192.168.144.0/24 + gateway: 192.168.144.1 + - name: internalapi + dnsDomain: internalapi.example.com + subnets: + - name: subnet1 # Central site + allocationRanges: + - end: 172.17.0.250 + start: 172.17.0.100 + cidr: 172.17.0.0/24 + vlan: 20 + - name: internalapidcn1 # DCN1 site + allocationRanges: + - end: 172.17.10.250 + start: 172.17.10.100 + cidr: 172.17.10.0/24 + vlan: 30 + - name: internalapidcn2 # DCN2 site + allocationRanges: + - end: 172.17.20.250 + start: 172.17.20.100 + cidr: 172.17.20.0/24 + vlan: 40 +---- + +* Each network defines multiple subnets, one for each site. +* Each site uses unique VLAN IDs. In this example, central uses VLANs 20-23, DCN1 uses VLANs 30-33, and DCN2 uses VLANs 40-43. +* The subnet naming convention typically uses `subnet1` for the central site and site-specific names like `internalapidcn1` for edge sites. + +Because the sites are geopgraphically distributed, each site requires its own provider network (physnet). The {networking_first_ref} must be configured to recognize all physnets. + +.Example Neutron ML2 configuration for multiple physnets +[source,yaml] +---- +[ml2_type_vlan] +network_vlan_ranges = leaf0:1:1000,leaf1:1:1000,leaf2:1:1000 + +[neutron] +physnets = leaf0,leaf1,leaf2 +---- + +* `leaf0` corresponds to the central site. +* `leaf1` corresponds to the DCN1 site. +* `leaf2` corresponds to the DCN2 site. + +When you create routed provider networks in {rhos_acro}, you create network segments that map to these physnets: + +* Segment for central: `physnet=leaf0`, subnet=192.168.122.0/24 +* Segment for DCN1: `physnet=leaf1`, subnet=192.168.133.0/24 +* Segment for DCN2: `physnet=leaf2`, subnet=192.168.144.0/24 + +[role="_additional-resources"] +.Additional resources +* xref:configuring-control-plane-networking-for-spine-leaf_troubleshooting-hsm[Configuring control plane networking for spine-leaf topologies] diff --git a/docs_user/modules/con_adoption-limitations.adoc b/docs_user/modules/con_adoption-limitations.adoc index 3d627764e..1e1b767a1 100644 --- a/docs_user/modules/con_adoption-limitations.adoc +++ b/docs_user/modules/con_adoption-limitations.adoc @@ -10,7 +10,7 @@ Technology Preview:: + The following features are Technology Previews and have not been tested within the context of the {rhos_long} adoption: + -* FC-based drivers for {block_storage_first_ref} +* {key_manager_first_ref} adoption with Proteccio hardware security module (HSM) integration + The following {compute_service_first_ref} features are Technology Previews: + @@ -29,7 +29,7 @@ Unsupported features:: + The adoption process does not support the following features: + -* Distributed Compute node architecture (DCN) +* Distributed Compute Node (DCN) architecture with storage services at remote or edge sites * DNS-as-a-service (designate) * {loadbalancer_first_ref} * Adopting Border Gateway Protocol (BGP) environments to the {rhos_acro} data plane diff --git a/docs_user/modules/con_adoption-process-overview.adoc b/docs_user/modules/con_adoption-process-overview.adoc index b6ef38c50..bba49316c 100644 --- a/docs_user/modules/con_adoption-process-overview.adoc +++ b/docs_user/modules/con_adoption-process-overview.adoc @@ -33,3 +33,4 @@ Post-adoption tasks:: For more information about updating your environment, see link:https://docs.redhat.com/en/documentation/red_hat_openstack_services_on_openshift/18.0/html/updating_your_environment_to_the_latest_maintenance_release/index[Updating your environment to the latest maintenance release]. * Optional: Verify that you migrated all services from the Controller nodes, and then power off the nodes. If any services are still running in the Controller nodes, such as Open Virtual Networking (ML2/OVN), {object_storage_first_ref}, or {Ceph}, do not power off the nodes. * If you enabled the high availability for Compute instances (Instance HA) service, remove the Pacemaker components from your Compute nodes. For more information, see xref:enabling-high-availability-for-instances_data-plane[Enabling the high availability for Compute instances service]. +* Enable TLS Everywhere (TLS-e). For more information about enabling TLS-e after completing the adoption, see link:https://docs.redhat.com/en/documentation/red_hat_openstack_services_on_openshift/18.0/html-single/configuring_security_services/index#assembly_enabling-TLS-on-a-deployed-RHOSO-environment[Enabling TLS on a deployed RHOSO environment] in _Configuring security services_. diff --git a/docs_user/modules/con_dcn-adoption-overview.adoc b/docs_user/modules/con_dcn-adoption-overview.adoc new file mode 100644 index 000000000..2543d7b37 --- /dev/null +++ b/docs_user/modules/con_dcn-adoption-overview.adoc @@ -0,0 +1,75 @@ +:_mod-docs-content-type: CONCEPT +[id="dcn-adoption-overview_{context}"] + += Overview of Distributed Compute Node adoption + +[role="_abstract"] +The process to adopt Distributed Compute Node (DCN) deployment from {rhos_prev_long} ({OpenStackShort}) to {rhos_long} requires additional adoption tasks: + +* You must map a multi-stack deployment to multiple node sets. +* You must map additional networking configurations. + +Multi-stack to multi-node set mapping:: In {OpenStackPreviousInstaller} deployments, DCN environments use multiple Heat stacks: ++ +** The Central stack is templating for Controllers and central Compute nodes. +** An edge stack is templating for Edge Compute nodes in a stack. There is one stack per DCN site. ++ +When you perform an adoption, map {OpenStackPreviousInstaller} stacks to `OpenStackDataPlaneNodeSet` custom resources (CRs): ++ +.Mapping {OpenStackPreviousInstaller} stacks to {rhos_acro} nodesets +[options="header"] +|=== +| {OpenStackPreviousInstaller} stack | {rhos_acro} nodeset | Availability zone +| Central stack (Compute role) | `openstack-edpm` or `openstack-cell1` | az-central +| DCN1 stack (ComputeDcn1 role) | `openstack-edpm-dcn1` or `openstack-cell1-dcn1` | az-dcn1 +| DCN2 stack (ComputeDcn2 role) | `openstack-edpm-dcn2` or `openstack-cell1-dcn2` | az-dcn2 +|=== ++ +[NOTE] +==== +Keep all node sets in the same Nova cell to maintain unified scheduling through a shared cell. The default cell is `cell1`. +==== + +Key differences from standard adoption:: The following table summarizes the differences between standard adoption and DCN adoption: ++ +.Comparison of standard and DCN adoption +[options="header"] +|=== +| Aspect | Standard adoption | DCN adoption +| Director stacks | Single stack | Multiple stacks (central + edge sites) +| Network topology | Flat L2 networks | Routed L3 networks with multiple subnets +| Data plane node sets | Single node set | Multiple node sets (one per site minimum) +| Network routes | Usually not required | Required for inter-site connectivity +| Physnets | Single physnet (e.g., `datacentre`) | Multiple physnets (e.g., `leaf0`, `leaf1`, `leaf2`) +| Availability zones | Often single AZ | Multiple AZs (one per site) +| OVN bridge mappings | Single mapping | Site-specific mappings +| Provider networks | Single segment | Multi-segment routed provider networks +|=== + +Requirements for DCN adoption:: Before adopting a DCN deployment, ensure you have: ++ +** Network topology information for all sites (IP ranges, VLANs, gateways) +** Inter-site routing configuration (routes between site subnets) +** Mapping of {OpenStackPreviousInstaller} roles to availability zones +** OVN bridge mapping configuration for each site + +[IMPORTANT] +==== +The adoption of the control plane must complete before adopting any data plane nodes. However, once the control plane is adopted, the edge site data plane adoptions can proceed in parallel with the central site data plane adoption. +==== + +DCN Adoption workflow overview:: The adoption of a Distributed Compute Node (DCN) deployment from {rhos_prev_long} ({OpenStackShort}) to {rhos_long} ++ +. **Control plane adoption**: Adopt all control plane services from the central {OpenStackPreviousInstaller} stack to the {rhos_acro} control plane. This is identical to standard adoption. +. **Network configuration**: Configure multi-subnet `NetConfig` and `NetworkAttachmentDefinition` CRs to support all site networks. +. **Data plane node set creation**: Create separate `OpenStackDataPlaneNodeSet` CRs for each site, each with site-specific network configurations: ++ +** Network subnet references +** OVN bridge mappings (physnets) +** Inter-site routing configuration +. **Data plane deployment**: Deploy all node sets. The edge site node sets can be deployed in parallel after the central site control plane is adopted. + + +[role="_additional-resources"] +.Additional resources +* xref:adopting-spine-leaf-networks_planning[Configuring spine-leaf networks for the {rhos_long_noacro} deployment] diff --git a/docs_user/modules/con_preventing-config-loss-when-using-oc-patch.adoc b/docs_user/modules/con_preventing-config-loss-when-using-oc-patch.adoc new file mode 100644 index 000000000..c9e7f6028 --- /dev/null +++ b/docs_user/modules/con_preventing-config-loss-when-using-oc-patch.adoc @@ -0,0 +1,18 @@ +:_mod-docs-content-type: CONCEPT +[id="preventing-config-loss-when-using-oc-patch_{context}"] + += Preventing configuration loss when using the `oc patch` command + +[role="_abstract"] +When you use the `oc patch` command to modify a resource, the changes are applied directly to the live object in your OpenShift cluster. If you later edit the custom resource (CR) file for the resource and apply the updates by using `oc apply -f `, your previous patched changes are overwritten and lost from the resource. + +To prevent loss of configuration, you can use the `--patch-file` option to configure the patch and retain patch files. Alternatively, you can export your `openstackcontrolplane` CR after the patch is applied: + +---- +$ oc get -o yaml > .yaml +---- + +For example: +---- +$ oc get OpenStackControlPlane openstack-control-plane -o yaml > openstack_control_plane.yaml +---- diff --git a/docs_user/modules/proc_adopting-compute-services-to-the-data-plane.adoc b/docs_user/modules/proc_adopting-compute-services-to-the-data-plane.adoc index 5406fe19d..c205e11d7 100644 --- a/docs_user/modules/proc_adopting-compute-services-to-the-data-plane.adoc +++ b/docs_user/modules/proc_adopting-compute-services-to-the-data-plane.adoc @@ -369,10 +369,10 @@ $ for CELL in $(echo $RENAMED_CELLS); do ip_api="${ref_api}['$compute']" cat >> computes-$CELL << EOF ${compute}: - *hostName: $compute* + hostName: $compute ansible: ansibleHost: $compute - *networks:* + networks: - defaultRoute: true fixedIP: ${!ip} name: ctlplane @@ -393,13 +393,13 @@ EOF apiVersion: dataplane.openstack.org/v1beta1 kind: OpenStackDataPlaneNodeSet metadata: - *name: openstack-$CELL* + name: openstack-$CELL spec: - *tlsEnabled: false* + tlsEnabled: false networkAttachments: - ctlplane preProvisioned: true - *services*: + services: ifeval::["{build}" == "downstream"] - redhat endif::[] @@ -494,7 +494,7 @@ endif::[] # # These vars are for the network config templates themselves and are # considered EDPM network defaults. - *neutron_physical_bridge_name: br-ctlplane* + neutron_physical_bridge_name: br-ctlplane neutron_public_interface_name: eth0 # edpm_nodes_validation @@ -502,7 +502,7 @@ endif::[] edpm_nodes_validation_validate_gateway_icmp: false # edpm ovn-controller configuration - *edpm_ovn_bridge_mappings: * + edpm_ovn_bridge_mappings: edpm_ovn_bridge: br-int edpm_ovn_encap_type: geneve ovn_monitor_all: true @@ -551,8 +551,8 @@ endif::[] edpm_ovs_packages: - openvswitch3.3 edpm_default_mounts: - - *path: /dev/hugepages* - *opts: pagesize=* + - path: /dev/hugepages + opts: pagesize= fstype: hugetlbfs group: hugetlbfs nodes: diff --git a/docs_user/modules/proc_adopting-the-block-storage-service.adoc b/docs_user/modules/proc_adopting-the-block-storage-service.adoc index 5e8ac9a6f..a1ac480fb 100644 --- a/docs_user/modules/proc_adopting-the-block-storage-service.adoc +++ b/docs_user/modules/proc_adopting-the-block-storage-service.adoc @@ -13,7 +13,8 @@ To adopt a {OpenStackPreviousInstaller}-deployed {block_storage_first_ref}, crea * You have prepared the {rhocp_long} nodes where the volume and backup services run. For more information, see xref:openshift-preparation-for-block-storage-adoption_storage-requirements[{OpenShiftShort} preparation for {block_storage} adoption]. * The {block_storage_first_ref} is stopped. * The service databases are imported into the control plane MariaDB. -* The {identity_service_first_ref} and {key_manager_first_ref} are adopted. +* The {identity_service_first_ref} is adopted. +* If your {rhos_prev_long} {rhos_prev_ver} deployment included the {key_manager_first_ref}, the {key_manager} is adopted. * The Storage network is correctly configured on the {OpenShiftShort} cluster. * The contents of `cinder.conf` file. Download the file so that you can access it locally: + diff --git a/docs_user/modules/proc_adopting-the-loadbalancer-service.adoc b/docs_user/modules/proc_adopting-the-loadbalancer-service.adoc index 75010682e..1f383a533 100644 --- a/docs_user/modules/proc_adopting-the-loadbalancer-service.adoc +++ b/docs_user/modules/proc_adopting-the-loadbalancer-service.adoc @@ -151,58 +151,6 @@ $ openstack endpoint list --service load-balancer +----------------------------------+-----------+--------------+---------------+---------+-----------+---------------------------------------------------+ ---- -.Post-adoption cleanup +.Next steps -Before running the post-adoption cleanup, you can ensure that the connectivty between the new control plane and the adopted Compute nodes is functional by creating a new load balancer and checking that its `provisioning_status` becomes `ACTIVE`. - ----- -$ alias openstack="oc exec -t openstackclient -- openstack" -$ openstack loadbalancer create --vip-subnet-id public-subnet --name lb-post-adoption --wait ----- - -After you complete the data plane adoption, perform the following cleanup steps to upgrade existing load balancers and remove old resources. - -. Trigger a failover for all existing load balancers to upgrade the amphorae virtual machines to use the new image and to establish connectivity with the new control plane: -+ ----- -$ openstack loadbalancer list -f value -c id | \ - xargs -r -n1 -P4 ${BASH_ALIASES[openstack]} loadbalancer failover --wait ----- - -. Delete old flavors that were migrated to the new control plane: -+ ----- -$ openstack flavor delete octavia_65 -# The following flavors might not exist in OSP 17.1 deployments -$ openstack flavor show octavia_amphora-mvcpu-ha && \ - openstack flavor delete octavia_amphora-mvcpu-ha -$ openstack loadbalancer flavor show octavia_amphora-mvcpu-ha && \ - openstack loadbalancer flavor delete octavia_amphora-mvcpu-ha -$ openstack loadbalancer flavorprofile show octavia_amphora-mvcpu-ha_profile && \ - openstack loadbalancer flavorprofile delete octavia_amphora-mvcpu-ha_profile ----- -+ -[NOTE] -Some flavors might still be used by load balancers and cannot be deleted. - -. Delete the old management network and its ports: -+ ----- -$ for net_id in $(openstack network list -f value -c ID --name lb-mgmt-net); do \ - desc=$(openstack network show "$net_id" -f value -c description); \ - [ -z "$desc" ] && WALLABY_LB_MGMT_NET_ID="$net_id" ; \ - done -$ for id in $(openstack port list --network "$WALLABY_LB_MGMT_NET_ID" -f value -c ID); do \ - openstack port delete "$id" ; \ - done -$ openstack network delete "$WALLABY_LB_MGMT_NET_ID" ----- - -. Verify that only one `lb-mgmt-net` and one `lb-mgmt-subnet` exists: -+ ----- -$ openstack network list | grep lb-mgmt-net -| fe470c29-0482-4809-9996-6d636e3feea3 | lb-mgmt-net | 6a881091-097d-441c-937b-5a23f4f243b7 | -$ openstack subnet list | grep lb-mgmt-subnet -| 6a881091-097d-441c-937b-5a23f4f243b7 | lb-mgmt-subnet | fe470c29-0482-4809-9996-6d636e3feea3 | 172.24.0.0/16 | ----- +After you complete the data plane adoption, you must upgrade existing load balancers and remove old resources. For more information, see xref:performing-post-adoption-cleanup-of-load-balancers_data-plane[Post-adoption tasks for the {loadbalancer_service}]. diff --git a/docs_user/modules/proc_configuring-control-plane-networking-for-spine-leaf.adoc b/docs_user/modules/proc_configuring-control-plane-networking-for-spine-leaf.adoc new file mode 100644 index 000000000..4ba45bd6a --- /dev/null +++ b/docs_user/modules/proc_configuring-control-plane-networking-for-spine-leaf.adoc @@ -0,0 +1,313 @@ +:_mod-docs-content-type: PROCEDURE +[id="configuring-control-plane-networking-for-spine-leaf_{context}"] + += Configuring control plane networking for spine-leaf topologies + +[role="_abstract"] +If you are adopting a spine-leaf or Distributed Compute Node (DCN) deployment, update the control plane networking for communication across sites. Add subnets for remote sites to your existing `NetConfig` custom resource (CR) and update `NetworkAttachmentDefinition` CRs with routes to enable connectivity between the central control plane and remote sites. + +.Prerequisites + +* You have deployed the {rhos_long} control plane. +* You have configured a `NetConfig` CR for the central site. For more information, see xref:configuring-isolated-networks_configuring-network[Configuring isolated networks]. +* You have the network topology information for all remote sites, including: +** IP address ranges for each service network at each site +** VLAN IDs for each service network at each site +** Gateway addresses for inter-site routing + +.Procedure + +. Update your existing `NetConfig` CR to add subnets for each remote site. Each service network must include a subnet for the central site and each remote site. Use unique VLAN IDs for each site. For example: ++ +* Central site: VLANs 20-23 +* Edge site 1: VLANs 30-33 +* Edge site 2: VLANs 40-43 ++ +[source,yaml] +---- +apiVersion: network.openstack.org/v1beta1 +kind: NetConfig +metadata: + name: netconfig +spec: + networks: + - name: ctlplane + dnsDomain: ctlplane.example.com + subnets: + - name: + allocationRanges: + - end: 192.168.122.120 + start: 192.168.122.100 + cidr: 192.168.122.0/24 + gateway: 192.168.122.1 + - name: + allocationRanges: + - end: 192.168.133.120 + start: 192.168.133.100 + cidr: 192.168.133.0/24 + gateway: 192.168.133.1 + - name: + allocationRanges: + - end: 192.168.144.120 + start: 192.168.144.100 + cidr: 192.168.144.0/24 + gateway: 192.168.144.1 + - name: internalapi + dnsDomain: internalapi.example.com + subnets: + - name: subnet1 + allocationRanges: + - end: 172.17.0.250 + start: 172.17.0.100 + cidr: 172.17.0.0/24 + vlan: 20 + - name: internalapisite1 + allocationRanges: + - end: 172.17.10.250 + start: 172.17.10.100 + cidr: 172.17.10.0/24 + vlan: 30 + - name: internalapisite2 + allocationRanges: + - end: 172.17.20.250 + start: 172.17.20.100 + cidr: 172.17.20.0/24 + vlan: 40 + - name: storage + dnsDomain: storage.example.com + subnets: + - name: subnet1 + allocationRanges: + - end: 172.18.0.250 + start: 172.18.0.100 + cidr: 172.18.0.0/24 + vlan: 21 + - name: storagesite1 + allocationRanges: + - end: 172.18.10.250 + start: 172.18.10.100 + cidr: 172.18.10.0/24 + vlan: 31 + - name: storagesite2 + allocationRanges: + - end: 172.18.20.250 + start: 172.18.20.100 + cidr: 172.18.20.0/24 + vlan: 41 + - name: tenant + dnsDomain: tenant.example.com + subnets: + - name: subnet1 + allocationRanges: + - end: 172.19.0.250 + start: 172.19.0.100 + cidr: 172.19.0.0/24 + vlan: 22 + - name: tenantsite1 + allocationRanges: + - end: 172.19.10.250 + start: 172.19.10.100 + cidr: 172.19.10.0/24 + vlan: 32 + - name: tenantsite2 + allocationRanges: + - end: 172.19.20.250 + start: 172.19.20.100 + cidr: 172.19.20.0/24 + vlan: 42 +---- ++ +where: + +``:: Specifies a user-defined subnet name for the central site subnet. +``:: Specifies a user-defined subnet for the first DCN edge site. +``:: Specifies a user-defined subnet for the second DCN edge site. ++ +[NOTE] +==== +You must have the `storagemgmt` network on OpenShift nodes when using DCN with Swift storage. It is not necessary when using Red Hat Ceph Storage. +==== + +. Update the `NetworkAttachmentDefinition` CR for the `internalapi` network to include routes to remote site subnets. These `routes` fields enable control plane pods attached to the `internalapi` network, such as OVN Southbound database, to communicate with Compute nodes at remote sites through the central site gateway, and are required for DCN: ++ +[source,yaml] +---- +apiVersion: k8s.cni.cncf.io/v1 +kind: NetworkAttachmentDefinition +metadata: + name: internalapi +spec: + config: | + { + "cniVersion": "0.3.1", + "name": "internalapi", + "type": "macvlan", + "master": "internalapi", + "ipam": { + "type": "whereabouts", + "range": "172.17.0.0/24", + "range_start": "172.17.0.30", + "range_end": "172.17.0.70", + "routes": [ + { "dst": "172.17.10.0/24", "gw": "172.17.0.1" }, + { "dst": "172.17.20.0/24", "gw": "172.17.0.1" } + ] + } + } +---- + +. Update the `NetworkAttachmentDefinition` CR for the `ctlplane` network to include routes to remote site subnets: ++ +[source,yaml] +---- +apiVersion: k8s.cni.cncf.io/v1 +kind: NetworkAttachmentDefinition +metadata: + name: ctlplane +spec: + config: | + { + "cniVersion": "0.3.1", + "name": "ctlplane", + "type": "macvlan", + "master": "ospbr", + "ipam": { + "type": "whereabouts", + "range": "192.168.122.0/24", + "range_start": "192.168.122.30", + "range_end": "192.168.122.70", + "routes": [ + { "dst": "192.168.133.0/24", "gw": "192.168.122.1" }, + { "dst": "192.168.144.0/24", "gw": "192.168.122.1" } + ] + } + } +---- + +. Update the `NetworkAttachmentDefinition` CR for the `storage` network to include routes to remote site subnets: ++ +[source,yaml] +---- +apiVersion: k8s.cni.cncf.io/v1 +kind: NetworkAttachmentDefinition +metadata: + name: storage +spec: + config: | + { + "cniVersion": "0.3.1", + "name": "storage", + "type": "macvlan", + "master": "storage", + "ipam": { + "type": "whereabouts", + "range": "172.18.0.0/24", + "range_start": "172.18.0.30", + "range_end": "172.18.0.70", + "routes": [ + { "dst": "172.18.10.0/24", "gw": "172.18.0.1" }, + { "dst": "172.18.20.0/24", "gw": "172.18.0.1" } + ] + } + } +---- + +. Update the `NetworkAttachmentDefinition` CR for the `tenant` network to include routes to remote site subnets: ++ +[source,yaml] +---- +apiVersion: k8s.cni.cncf.io/v1 +kind: NetworkAttachmentDefinition +metadata: + name: tenant +spec: + config: | + { + "cniVersion": "0.3.1", + "name": "tenant", + "type": "macvlan", + "master": "tenant", + "ipam": { + "type": "whereabouts", + "range": "172.19.0.0/24", + "range_start": "172.19.0.30", + "range_end": "172.19.0.70", + "routes": [ + { "dst": "172.19.10.0/24", "gw": "172.19.0.1" }, + { "dst": "172.19.20.0/24", "gw": "172.19.0.1" } + ] + } + } +---- ++ +[NOTE] +==== +Adjust the IP ranges, subnets, and gateway addresses in all NAD configurations to match your network topology. The `master` interface name must match the interface on the OpenShift nodes where the VLAN is configured. +==== + +. If you have already deployed OVN services, restart the OVN Southbound database pods to pick up the new routes: ++ +---- +$ oc delete pod -l service=ovsdbserver-sb +---- ++ +The pods are automatically recreated with the updated network configuration. + +. Configure the {networking_first_ref} to recognize all site physnets. In the `OpenStackControlPlane` CR, ensure the Networking service configuration includes all physnets: ++ +[source,yaml] +---- +apiVersion: core.openstack.org/v1beta1 +kind: OpenStackControlPlane +metadata: + name: openstack +spec: + neutron: + template: + customServiceConfig: | + [ml2_type_vlan] + network_vlan_ranges = leaf0:1:1000,leaf1:1:1000,leaf2:1:1000 + [ovn] + ovn_emit_need_to_frag = false +---- ++ +where: + +leaf0:: Represents the physnet for the central site. +leaf1:: Represents the physnet for the first remote site. +leaf2:: Represents the physnet for the second remote site. ++ +[NOTE] +==== +Adjust the physnet names to match your {rhos_prev_long} deployment. Common conventions include `leaf0/leaf1/leaf2` or `datacentre/dcn1/dcn2`. +==== + +.Verification + +. Verify that the `NetConfig` CR is created with all subnets: ++ +---- +$ oc get netconfig netconfig -o yaml | grep -A2 "name: subnet1\|name: .*site" +---- + +. Verify that each `NetworkAttachmentDefinition` includes routes to remote site subnets: ++ +---- +for nad in ctlplane internalapi storage tenant; do + echo "=== $nad ===" + oc get net-attach-def $nad -o jsonpath='{.spec.config}' | jq '.ipam.routes +done +---- + +. After restarting OVN SB pods, verify they have routes to remote site subnets: ++ +---- +$ oc exec $(oc get pod -l service=ovsdbserver-sb -o name | head -1) -- ip route show | grep 172.17 +---- ++ +Sample output: ++ +---- +172.17.10.0/24 via 172.17.0.1 dev internalapi +172.17.20.0/24 via 172.17.0.1 dev internalapi +---- diff --git a/docs_user/modules/proc_configuring-data-plane-nodes.adoc b/docs_user/modules/proc_configuring-data-plane-nodes.adoc index e442f9e3f..36788fead 100644 --- a/docs_user/modules/proc_configuring-data-plane-nodes.adoc +++ b/docs_user/modules/proc_configuring-data-plane-nodes.adoc @@ -48,7 +48,10 @@ spec: cidr: 172.19.0.0/24 vlan: 22 ---- -* `spec.networks` specifies the `networks` composition. The `networks` composition must match the source cloud configuration to avoid data plane connectivity downtime. ++ +where: + +spec.networks:: Specifies the `networks` composition. The `networks` composition must match the source cloud configuration to avoid data plane connectivity downtime. . Optional: In the `NetConfig` CR, list multiple ranges for the `allocationRanges` field to exclude some of the IP addresses, for example, to accommodate IP addresses that are already consumed by the adopted environment: + diff --git a/docs_user/modules/proc_configuring-dcn-data-plane-nodesets.adoc b/docs_user/modules/proc_configuring-dcn-data-plane-nodesets.adoc new file mode 100644 index 000000000..d9f367aca --- /dev/null +++ b/docs_user/modules/proc_configuring-dcn-data-plane-nodesets.adoc @@ -0,0 +1,361 @@ +:_mod-docs-content-type: PROCEDURE +[id="configuring-dcn-data-plane-nodesets_{context}"] + += Configuring data plane node sets for DCN sites + +[role="_abstract"] +If you are adopting a Distributed Compute Node (DCN) deployment, you must create separate `OpenStackDataPlaneNodeSet` custom resources (CRs) for each site. Each node set requires site-specific configuration for network subnets, OVN bridge mappings, and inter-site routes. + +.Prerequisites + +* You have adopted the {rhos_prev_long} ({OpenStackShort}) control plane to {rhos_long}. +* You have configured control plane networking for your spine-leaf topology, including multi-subnet `NetConfig` and `NetworkAttachmentDefinition` CRs with routes to remote sites. For more information, see xref:configuring-control-plane-networking-for-spine-leaf_adopt-control-plane[Configuring control plane networking for spine-leaf topologies]. +* You have the network configuration information for each DCN site: +** IP addresses and hostnames for all Compute nodes +** VLAN IDs for each service network +** Gateway addresses for inter-site routing +* You have identified the OVN bridge mappings (physnets) for each site. + +.Procedure + +. Define the OVN bridge mappings for each site. Each site requires a unique physnet that maps to the local provider network bridge: ++ +.Example OVN bridge mappings +[options="header"] +|=== +| Site | OVN bridge mapping +| Central | `leaf0:br-ex` +| DCN1 | `leaf1:br-ex` +| DCN2 | `leaf2:br-ex` +|=== + +. Configure OVN for DCN sites. The default OVN controller configuration uses the Kubernetes ClusterIP (`ovsdbserver-sb.openstack.svc`), which is not routable from remote DCN sites. You must create a DCN-specific configuration that uses direct `internalapi` IP addresses. + +.. Get the OVN Southbound database `internalapi` IP addresses: ++ +---- +$ oc get pod -l service=ovsdbserver-sb -o jsonpath='{range .items[*]}{.metadata.annotations.k8s\.v1\.cni\.cncf\.io/network-status}{"\n"}{end}' | jq -r '.[] | select(.name=="openstack/internalapi") | .ips[0]' +---- ++ +Example output: ++ +---- +172.17.0.34 +172.17.0.35 +172.17.0.36 +---- + +.. Create a ConfigMap with the OVN SB direct IPs for DCN sites: ++ +[subs="+quotes"] +---- +$ oc apply -f - < + {{ ctlplane_host_routes }} + *- ip_netmask: 192.168.122.0/24* + *next_hop: 192.168.133.1* + - ip_netmask: 192.168.144.0/24 + next_hop: 192.168.133.1 + members: + - type: interface + name: nic1 + primary: true + {% for network in nodeset_networks %} + - type: vlan + vlan_id: {{ lookup('vars', networks_lower[network] ~ '_vlan_id') }} + addresses: + - ip_netmask: + {{ lookup('vars', networks_lower[network] ~ '_ip') }}/{{ lookup('vars', networks_lower[network] ~ '_cidr') }} + routes: + {{ lookup('vars', networks_lower[network] ~ '_host_routes') }} + {% if network == 'internalapi' %} + *- ip_netmask: 172.17.0.0/24* + *next_hop: 172.17.10.1* + - ip_netmask: 172.17.20.0/24 + next_hop: 172.17.10.1 + {% endif %} + {% if network == 'storage' %} + - ip_netmask: 172.18.0.0/24 + next_hop: 172.18.10.1 + - ip_netmask: 172.18.20.0/24 + next_hop: 172.18.10.1 + {% endif %} + {% if network == 'tenant' %} + - ip_netmask: 172.19.0.0/24 + next_hop: 172.19.10.1 + - ip_netmask: 172.19.20.0/24 + next_hop: 172.19.10.1 + {% endif %} + {% endfor %} + nodes: + dcn1-compute-0: + hostName: dcn1-compute-0.example.com + ansible: + ansibleHost: dcn1-compute-0.example.com + networks: + - defaultRoute: true + fixedIP: 192.168.133.100 + name: ctlplane + *subnetName: ctlplanedcn1* + - name: internalapi + *subnetName: internalapidcn1* + - name: storage + *subnetName: storagedcn1* + - name: tenant + *subnetName: tenantdcn1* +---- +* Replace `ovn` with `ovn-dcn` under spec:services. This ensures OVN controller connects to the OVN Southbound database using direct internalapi IPs instead of the unreachable ClusterIP. +* DCN1 uses the `leaf1` physnet, for its OVN bridge mapping under `spec:nodeTemplate:ansible:ansibleVars:edpm_ovn_bridge_mappings`. +* Inter-site routes must be added to the network configuration template. These routes enable DCN1 compute nodes to reach the central site (192.168.122.0/24) and other DCN sites (192.168.144.0/24 for DCN2). Similar routes are added for each service network (internalapi, storage, tenant). +* DCN1 nodes reference site-specific subnet names like `ctlplanedcn1` and `internalapidcn1`. These subnet names must match those defined in the `NetConfig` CR. + +. Repeat step 3 for all other DCN sites. Adjust site specific parameters: ++ +* The nodeset name, for example: `openstack-edpm-dcn2` +* The OVN bridge mapping, for example: `leaf2:br-ex` +* The subnet names, for example: `ctlplanedcn2`, and `internalapidcn2` +* The inter-site routes. The routes from DCN2 should point to the central site subnets and the DCN1 site subnets. +* The compute node definitions with site-appropriate IP addresses. + +. Deploy all nodesets by creating an `OpenStackDataPlaneDeployment` CR: ++ +[source,yaml] +---- +apiVersion: dataplane.openstack.org/v1beta1 +kind: OpenStackDataPlaneDeployment +metadata: + name: openstack-edpm-deployment +spec: + nodeSets: + - openstack-edpm + - openstack-edpm-dcn1 + - openstack-edpm-dcn2 +---- ++ +[NOTE] +==== +All nodesets can be deployed in parallel once the control plane adoption is complete. +==== + +. Wait for the deployment to complete: ++ +---- +$ oc wait --for condition=Ready openstackdataplanedeployment/openstack-edpm-deployment --timeout=40m +---- + +.Verification + +. Verify that all node sets reach the `Ready` status: ++ +---- +$ oc get openstackdataplanenodeset +NAME STATUS MESSAGE +openstack-edpm True Ready +openstack-edpm-dcn1 True Ready +openstack-edpm-dcn2 True Ready +---- + +. Verify that Compute services are running across all sites. Ensure that all `nova-compute` services show `State=up` for nodes in all availability zones: ++ +---- +$ oc exec openstackclient -- openstack compute service list +---- + +. Verify inter-site connectivity by checking routes on a DCN Compute node: ++ +---- +$ ssh dcn1-compute-0 ip route show | grep 172.17.0 +172.17.0.0/24 via 172.17.10.1 dev internalapi +---- + +. Test that DCN Compute nodes can reach the control plane: ++ +---- +$ ssh dcn1-compute-0 ping -c 3 172.17.0.30 +---- ++ +Replace `172.17.0.30` with an IP address of a control plane service on the internalapi network. diff --git a/docs_user/modules/proc_performing-post-adoption-cleanup-of-load-balancers.adoc b/docs_user/modules/proc_performing-post-adoption-cleanup-of-load-balancers.adoc new file mode 100644 index 000000000..814cbf543 --- /dev/null +++ b/docs_user/modules/proc_performing-post-adoption-cleanup-of-load-balancers.adoc @@ -0,0 +1,67 @@ +:_mod-docs-content-type: PROCEDURE +[id="performing-post-adoption-cleanup-of-load-balancers_{context}"] + += Post-adoption tasks for the {loadbalancer_service} + +[role="_abstract"] +If you adopted the {loadbalancer_first_ref}, after you complete the data plane adoption, you must perform the following tasks: + +* Upgrade the amphorae virtual machines to the new images. +* Remove obsolete resources from your existing load balancers. + +.Prerequisites +* You have adopted the {loadbalancer_service}. For more information, see xref:adopting-the-loadbalancer-service_troubleshooting-hsm[Adopting the {loadbalancer_service}]. + +.Procedure + +. Ensure that the connectivity between the new control plane and the adopted Compute nodes is functional by creating a new load balancer and checking that its `provisioning_status` becomes `ACTIVE`: ++ +---- +$ alias openstack="oc exec -t openstackclient -- openstack" +$ openstack loadbalancer create --vip-subnet-id public-subnet --name lb-post-adoption --wait +---- + +. Trigger a failover for all existing load balancers to upgrade the amphorae virtual machines to use the new image and to establish connectivity with the new control plane: ++ +---- +$ openstack loadbalancer list -f value -c id | \ + xargs -r -n1 -P4 ${BASH_ALIASES[openstack]} loadbalancer failover --wait +---- + +. Delete old flavors that were migrated to the new control plane: ++ +---- +$ openstack flavor delete octavia_65 +# The following flavors might not exist in OSP 17.1 deployments +$ openstack flavor show octavia_amphora-mvcpu-ha && \ + openstack flavor delete octavia_amphora-mvcpu-ha +$ openstack loadbalancer flavor show octavia_amphora-mvcpu-ha && \ + openstack loadbalancer flavor delete octavia_amphora-mvcpu-ha +$ openstack loadbalancer flavorprofile show octavia_amphora-mvcpu-ha_profile && \ + openstack loadbalancer flavorprofile delete octavia_amphora-mvcpu-ha_profile +---- ++ +[NOTE] +Some flavors might still be used by load balancers and cannot be deleted. + +. Delete the old management network and its ports: ++ +---- +$ for net_id in $(openstack network list -f value -c ID --name lb-mgmt-net); do \ + desc=$(openstack network show "$net_id" -f value -c description); \ + [ -z "$desc" ] && WALLABY_LB_MGMT_NET_ID="$net_id" ; \ + done +$ for id in $(openstack port list --network "$WALLABY_LB_MGMT_NET_ID" -f value -c ID); do \ + openstack port delete "$id" ; \ + done +$ openstack network delete "$WALLABY_LB_MGMT_NET_ID" +---- + +. Verify that only one `lb-mgmt-net` and one `lb-mgmt-subnet` exists: ++ +---- +$ openstack network list | grep lb-mgmt-net +| fe470c29-0482-4809-9996-6d636e3feea3 | lb-mgmt-net | 6a881091-097d-441c-937b-5a23f4f243b7 | +$ openstack subnet list | grep lb-mgmt-subnet +| 6a881091-097d-441c-937b-5a23f4f243b7 | lb-mgmt-subnet | fe470c29-0482-4809-9996-6d636e3feea3 | 172.24.0.0/16 | +---- diff --git a/docs_user/modules/proc_resolving-HSM-secret-creation-failures.adoc b/docs_user/modules/proc_resolving-HSM-secret-creation-failures.adoc new file mode 100644 index 000000000..64a0ac233 --- /dev/null +++ b/docs_user/modules/proc_resolving-HSM-secret-creation-failures.adoc @@ -0,0 +1,41 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-HSM-secret-creation-failures_{context}"] + += Resolving HSM secret creation failures + +[role="_abstract"] +If hardware security module (HSM) secrets cannot be created in the target environment, check whether you need to update the names of your secrets in your source configuration file. + +Example error: + +---- +TASK [Create HSM secrets in target environment] **** +fatal: [localhost]: FAILED! => { + "msg": "Failed to create secret proteccio-data" +} +---- + +.Procedure + +. Verify target environment access: ++ +[source,bash] +---- +$ export KUBECONFIG=/path/to/.kube/config +$ oc get secrets -n openstack +---- + +. Check if secrets already exist: ++ +[source,bash] +---- +$ oc get secret proteccio-data hsm-login -n openstack +---- + +. If secrets exist with different names, update the configuration variables: ++ +[source,yaml] +---- +proteccio_login_secret_name: "your-hsm-login-secret" +proteccio_client_data_secret_name: "your-proteccio-data-secret" +---- diff --git a/docs_user/modules/proc_resolving-adoption-verification-failures.adoc b/docs_user/modules/proc_resolving-adoption-verification-failures.adoc new file mode 100644 index 000000000..ce3a3d929 --- /dev/null +++ b/docs_user/modules/proc_resolving-adoption-verification-failures.adoc @@ -0,0 +1,41 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-adoption-verification-failures_{context}"] + += Resolving adoption verification failures + +[role="_abstract"] +If the secrets from the source environment are not accessible after adoption, verify that the database import completed successfully, test API connectivity, and check for schema adoption issues. + +Example error: + +---- +$ openstack secret list +# Returns empty list or HTTP 500 errors +---- + +.Procedure + +. Verify that the database import completed successfully: ++ +---- +$ oc exec openstack-galera-0 -n openstack -- mysql -u root -p barbican -e "SELECT COUNT(*) FROM secrets;" +---- ++ +where: + +``:: +Specifies your database password. + +. Check for schema adoption issues: ++ +---- +$ oc logs job.batch/barbican-db-sync -n openstack +---- + +. Test API connectivity: ++ +---- +$ oc exec openstackclient -n openstack -- curl -s -k -H "X-Auth-Token: $(openstack token issue -f value -c id)" https://barbican-internal.openstack.svc:9311/v1/secrets +---- + +. Verify that projects and users were adopted correctly, as secrets are project-scoped. diff --git a/docs_user/modules/proc_resolving-config-validation-failures.adoc b/docs_user/modules/proc_resolving-config-validation-failures.adoc new file mode 100644 index 000000000..0c7f1420d --- /dev/null +++ b/docs_user/modules/proc_resolving-config-validation-failures.adoc @@ -0,0 +1,33 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-configuration-validation-failures_{context}"] + += Resolving configuration validation failures + +[role="_abstract"] +If the adoption fails with validation errors about placeholder values, replace the placeholder values with your environment's configuration values. + +Example error: + +---- +TASK [Validate all required variables are set] **** +fatal: [localhost]: FAILED! => { + "msg": "Required variable proteccio_certs_path contains placeholder value." +} +---- + +.Procedure + +. Edit your hardware security module configuration in the Zuul job vars or CI framework configuration file. + +. Check the following key variables and replace all placeholder values with actual configuration values for your environment: ++ +---- +cifmw_hsm_password: +cifmw_barbican_proteccio_partition: +cifmw_barbican_proteccio_mkek_label: +cifmw_barbican_proteccio_hmac_label: +cifmw_hsm_proteccio_client_src: +cifmw_hsm_proteccio_conf_src: +---- + +. Verify that no placeholder values remain in your configuration. diff --git a/docs_user/modules/proc_resolving-connectivity-issues.adoc b/docs_user/modules/proc_resolving-connectivity-issues.adoc new file mode 100644 index 000000000..9d6b4cc99 --- /dev/null +++ b/docs_user/modules/proc_resolving-connectivity-issues.adoc @@ -0,0 +1,38 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-source-environment-connectivity-issues_{context}"] + += Resolving source environment connectivity issues + +[role="_abstract"] +If the adoption cannot connect to the source {rhos_prev_long} environment to extract the configuration, check your SSH connectivity to the source Controller node and update the configuration if needed. + +Example error: + +---- +TASK [detect source environment HSM configuration] **** +fatal: [localhost]: FAILED! => { + "msg": "SSH connection to source environment failed" +} +---- + +.Procedure + +. Verify SSH connectivity to the source Controller node: ++ +[source,bash] +---- +$ ssh -o StrictHostKeyChecking=no tripleo-admin@controller-0.ctlplane +---- + +. Update the `controller1_ssh` variable if needed: ++ +---- +$ controller1_ssh: "ssh -o StrictHostKeyChecking=no tripleo-admin@" +---- ++ +where: + +``:: +Specifies the IP address of your Controller node. + +. Ensure that the SSH keys are properly configured for passwordless access. diff --git a/docs_user/modules/proc_resolving-custom-image-pull-failures.adoc b/docs_user/modules/proc_resolving-custom-image-pull-failures.adoc new file mode 100644 index 000000000..c9d9e44ee --- /dev/null +++ b/docs_user/modules/proc_resolving-custom-image-pull-failures.adoc @@ -0,0 +1,46 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-custom-image-pull-failures_{context}"] + += Resolving custom image pull failures + +[role="_abstract"] +If Proteccio custom images fail to pull or start, verify image registry access, image pull secrets, and registry authentication. + +Example error: + +---- +Failed to pull image "": rpc error +Pod has unbound immediate PersistentVolumeClaims +---- + +.Procedure + +. Verify image registry access: ++ +---- +$ podman pull +---- ++ +where: + +``:: +Specifies your custom pod image and the image tag. + + +. Check image pull secrets and registry authentication: ++ +---- +$ oc get secrets -n openstack | grep pull +$ oc describe pod -n openstack +---- ++ +where: + +``:: +Specifies your Barbican pod name. + +. Verify that the `OpenStackVersion` resource was applied correctly: ++ +---- +$ oc get openstackversion openstack -n openstack -o yaml +---- diff --git a/docs_user/modules/proc_resolving-custom-image-registry-issues.adoc b/docs_user/modules/proc_resolving-custom-image-registry-issues.adoc new file mode 100644 index 000000000..b8681202b --- /dev/null +++ b/docs_user/modules/proc_resolving-custom-image-registry-issues.adoc @@ -0,0 +1,54 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-custom-image-registry-issues_{context}"] + += Resolving custom image registry issues + +[role="_abstract"] +If custom Barbican images cannot be pushed to or pulled from the configured registry, you can verify the authentication, test image push permissions, and then update the configuration as needed. + +Example error: + +---- +TASK [Create Proteccio-enabled Barbican images] **** +fatal: [localhost]: FAILED! => { + "msg": "Failed to push image to registry" +} +---- + +.Procedure + +. Verify registry authentication: ++ +[source,bash] +---- +$ podman login +---- ++ +where: + +``:: +Specifies the URL of your configured registry. + +. Test image push permissions: ++ +[source,bash] +---- +$ podman tag hello-world //test:latest +$ podman push //test:latest +---- ++ +where: + +``:: +Specifies the name of your registry server. +``:: +Specifies the namespace of your container image. + +. Update registry configuration variables if needed: ++ +[source,yaml] +---- +cifmw_update_containers_registry: "your-registry:5001" +cifmw_update_containers_org: "your-namespace" +cifmw_image_registry_verify_tls: false +---- diff --git a/docs_user/modules/proc_resolving-database-import-failures.adoc b/docs_user/modules/proc_resolving-database-import-failures.adoc new file mode 100644 index 000000000..fc112fed5 --- /dev/null +++ b/docs_user/modules/proc_resolving-database-import-failures.adoc @@ -0,0 +1,39 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-database-import-failures_{context}"] + += Resolving database import failures + +[role="_abstract"] +If the source database export or import fails, check the the source Galera container, database connectivity, and the source {key_manager_first_ref} configuration. + +Example error: + +---- +Error: no container with name or ID "galera-bundle-podman-0" found +mysqldump: Got error: 1045: "Access denied for user 'barbican'@'localhost'" +---- + +.Procedure + +. Verify that the source Galera container is running: ++ +---- +$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo podman ps | grep galera"' +---- + +. Test database connectivity with the extracted credentials: ++ +---- +$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo podman exec galera-bundle-podman-0 mysql -u barbican -p -e \"SELECT 1;\""' +---- ++ +where: + +``:: +Specifies your database password. + +. Check the source {key_manager} configuration for the correct database password: ++ +---- +$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo grep connection /var/lib/config-data/puppet-generated/barbican/etc/barbican/barbican.conf"' +---- diff --git a/docs_user/modules/proc_resolving-database-migration-issues.adoc b/docs_user/modules/proc_resolving-database-migration-issues.adoc new file mode 100644 index 000000000..103fb3bd4 --- /dev/null +++ b/docs_user/modules/proc_resolving-database-migration-issues.adoc @@ -0,0 +1,35 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-database-migration-issues_{context}"] + += Resolving database migration issues + +[role="_abstract"] +If hardware security module (HSM) metadata is not preserved during database migration, check the database logs for any errors and verify that the source database includes the HSM secrets. + +Example error: + +---- +TASK [Verify database migration preserves HSM references] **** +ok: [localhost] => { + "msg": "HSM secrets found in migrated database: 0" +} +---- + +.Procedure + +. Verify that the source database contains the HSM secrets: ++ +[source,bash] +---- +$ ssh tripleo-admin@controller-0.ctlplane \ + "sudo mysql barbican -e 'SELECT COUNT(*) FROM secret_store_metadata WHERE key=\"plugin_name\" AND value=\"PKCS11\";'" +---- + +. Check the database migration logs for errors: ++ +[source,bash] +---- +$ oc logs deployment/barbican-api | grep -i migration +---- + +. If the migration failed, restore the database from backup and retry. diff --git a/docs_user/modules/proc_resolving-hsm-backend-detection-failures.adoc b/docs_user/modules/proc_resolving-hsm-backend-detection-failures.adoc new file mode 100644 index 000000000..17273f599 --- /dev/null +++ b/docs_user/modules/proc_resolving-hsm-backend-detection-failures.adoc @@ -0,0 +1,37 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-HSM-back-end-detection-failures_{context}"] + +== Resolving HSM back-end detection failures + +[role="_abstract"] +If the adoption role cannot detect hardware security module (HSM) configuration in the source environment, you must force the HSM adoption. + +Example error: + +---- +TASK [detect source environment HSM configuration] **** +ok: [localhost] => { + "msg": "No HSM configuration found - using standard adoption" +} +---- + +.Procedure + +. Manually verify that the HSM configuration exists in the source environment: ++ +[source,bash] +---- +$ ssh tripleo-admin@controller-0.ctlplane \ + "sudo grep -A 10 '\[p11_crypto_plugin\]' \ + /var/lib/config-data/puppet-generated/barbican/etc/barbican/barbican.conf" +---- + +. If HSM is configured but not detected, force HSM adoption by setting the `barbican_hsm_enabled` variable: ++ +[source,yaml] +---- +# In your Zuul job vars or CI framework configuration +barbican_hsm_enabled: true +---- ++ +This configuration ensures that the `barbican_adoption` role uses the HSM-enabled patch for {key_manager_first_ref} deployment. diff --git a/docs_user/modules/proc_resolving-hsm-certificate-mounting-issues.adoc b/docs_user/modules/proc_resolving-hsm-certificate-mounting-issues.adoc new file mode 100644 index 000000000..b079237a0 --- /dev/null +++ b/docs_user/modules/proc_resolving-hsm-certificate-mounting-issues.adoc @@ -0,0 +1,34 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-HSM-certificate-mounting-issues_{context}"] + += Resolving HSM certificate mounting issues + +[role="_abstract"] +If Proteccio client certificates are not properly mounted in pods, check the secret creation and ensure that the {key_manager_first_ref} configuration includes the correct volume mounts. + +Example error: + +---- +$ oc exec -c barbican-api -- ls -la /etc/proteccio/ +ls: cannot access '/etc/proteccio/': No such file or directory +---- + +.Procedure + +. Verify that the `proteccio-data` secret was created correctly: ++ +---- +$ oc describe secret proteccio-data -n openstack +---- + +. Check that the secret contains the expected files: ++ +---- +$ oc get secret proteccio-data -n openstack -o yaml +---- + +. Verify that the {key_manager} configuration includes the correct volume mounts: ++ +---- +$ oc get barbican barbican -n openstack -o yaml | grep -A10 pkcs11 +---- diff --git a/docs_user/modules/proc_resolving-missing-HSM-file-prerequisites.adoc b/docs_user/modules/proc_resolving-missing-HSM-file-prerequisites.adoc new file mode 100644 index 000000000..f3464f1f1 --- /dev/null +++ b/docs_user/modules/proc_resolving-missing-HSM-file-prerequisites.adoc @@ -0,0 +1,39 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-missing-HSM-file-prerequisites_{context}"] + += Resolving missing HSM file prerequisites + +[role="_abstract"] +If the adoption fails because hardware security module (HSM) certificates or client software cannot be found, update your configuration to point to the files in their specific locations. + +Example error: + +---- +TASK [Validate Proteccio prerequisites exist] **** +fatal: [localhost]: FAILED! => { + "msg": "Proteccio client ISO not found: /opt/proteccio/Proteccio3.06.05.iso" +} +---- + +.Procedure + +. Verify that all required HSM files are accessible from the configured URLs. For example: ++ +[source,bash] +---- +$ curl -I https://your-server/path/to/Proteccio3.06.05.iso +$ curl -I https://your-server/path/to/proteccio.rc +$ curl -I https://your-server/path/to/client.crt +$ curl -I https://your-server/path/to/client.key +---- + +. If the files are in different locations, update the URL variables in your configuration. For example: ++ +---- +cifmw_hsm_proteccio_client_src: "https://correct-server/path/to/Proteccio3.06.05.iso" +cifmw_hsm_proteccio_conf_src: "https://correct-server/path/to/proteccio.rc" +cifmw_hsm_proteccio_client_crt_src: "https://correct-server/path/to/client.crt" +cifmw_hsm_proteccio_client_key_src: "https://correct-server/path/to/client.key" +---- + +. Check the network connectivity and authentication to ensure that the URLs are accessible from the CI environment. diff --git a/docs_user/modules/proc_resolving-performance-and-connectivity-issues.adoc b/docs_user/modules/proc_resolving-performance-and-connectivity-issues.adoc new file mode 100644 index 000000000..eddaed97c --- /dev/null +++ b/docs_user/modules/proc_resolving-performance-and-connectivity-issues.adoc @@ -0,0 +1,32 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-performance-and-connectivity-issues_{context}"] + += Resolving performance and connectivity issues + +[role="_abstract"] +If the hardware security module (HSM) operations are slow or fail intermittently, check the HSM connectivity and monitor the HSM server logs. + +.Procedure + +. Test HSM connectivity from {key_manager_first_ref} pods: ++ +[source,bash] +---- +$ oc exec barbican-api-xyz -- pkcs11-tool --module /usr/lib64/libnethsm.so --list-slots +---- + +. Check HSM server connectivity: ++ +[source,bash] +---- +$ oc exec barbican-api-xyz -- nc -zv +---- ++ +where: + +``:: +Specifies the IP address of the HSM server. +``:: +Specifies the port of your HSM server. + +. Monitor HSM server logs for authentication or capacity issues. diff --git a/docs_user/modules/proc_resolving-prerequisite-validation-failures.adoc b/docs_user/modules/proc_resolving-prerequisite-validation-failures.adoc new file mode 100644 index 000000000..73a46d1d3 --- /dev/null +++ b/docs_user/modules/proc_resolving-prerequisite-validation-failures.adoc @@ -0,0 +1,38 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-prerequisite-validation-failures_{context}"] + += Resolving prerequisite validation failures + +[role="_abstract"] +If the adoption script fails during the prerequisites check, verify that your configuration includes all the required Proteccio files and that the HSM Ansible role is available. + +Example error: + +---- +ERROR: Required file proteccio_files/YOUR_CERT_FILE not found +ERROR: Cannot connect to OpenShift cluster +ERROR: Proteccio HSM Ansible role not found +---- + +.Procedure + +. Verify that all required Proteccio files are present: ++ +---- +$ ls -la /path/to/your/proteccio_files/ +---- ++ +Ensure that your configured certificate files, private key, HSM certificate file, and configuration file exist as specified in your `proteccio_required_files` configuration. + +. Test OpenShift cluster connectivity: ++ +---- +$ oc cluster-info +$ oc get pods -n openstack +---- + +. Verify that the HSM Ansible role is available: ++ +---- +$ ls -la /path/to/your/roles/ansible-role-rhoso-proteccio-hsm/ +---- diff --git a/docs_user/modules/proc_resolving-service-startup-failures.adoc b/docs_user/modules/proc_resolving-service-startup-failures.adoc new file mode 100644 index 000000000..c915be88c --- /dev/null +++ b/docs_user/modules/proc_resolving-service-startup-failures.adoc @@ -0,0 +1,38 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-service-startup-failures_{context}"] + += Resolving service startup failures + +[role="_abstract"] +If the {key_manager_first_ref} services fail to start after the hardware security module (HSM) configuration is applied, check the configuration in the pod. + +Example error: + +---- +$ oc get pods -l service=barbican +NAME READY STATUS RESTARTS AGE +barbican-api-xyz 0/1 Error 0 2m +---- + +.Procedure + +. Check pod logs for HSM connectivity issues: ++ +[source,bash] +---- +$ oc logs barbican-api-xyz +---- + +. Verify HSM library is accessible: ++ +[source,bash] +---- +$ oc exec barbican-api-xyz -- ls -la /usr/lib64/libnethsm.so +---- + +. Check HSM configuration in the pod: ++ +[source,bash] +---- +$ oc exec barbican-api-xyz -- cat /etc/proteccio/proteccio.rc +---- diff --git a/docs_user/modules/proc_resolving-ssh-connection-failures.adoc b/docs_user/modules/proc_resolving-ssh-connection-failures.adoc new file mode 100644 index 000000000..d90d8b9a1 --- /dev/null +++ b/docs_user/modules/proc_resolving-ssh-connection-failures.adoc @@ -0,0 +1,31 @@ +:_mod-docs-content-type: PROCEDURE +[id="resolving-SSH-connection-failures_{context}"] + += Resolving SSH connection failures to the source environment + +[role="_abstract"] +If you cannot connect to the source {OpenStackPreviousInstaller} environment, verify your SSH key access and test the SSH commands that the adoption uses. + +Example error: + +---- +Warning: Permanently added 'YOUR_UNDERCLOUD_HOST' (ED25519) to the list of known hosts. +Permission denied (publickey). +---- + +.Procedure + +. Verify SSH key access to the undercloud: ++ +---- +$ ssh YOUR_UNDERCLOUD_HOST echo "Connection test" +---- + +. Test the specific SSH commands used by the adoption: ++ +---- +$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack bash -lc "echo test"' +$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "echo test"' +---- + +. If the connection fails, verify the SSH configuration and ensure that the undercloud hostname resolves correctly. diff --git a/docs_user/modules/proc_rolling-back-the-hsm-adoption.adoc b/docs_user/modules/proc_rolling-back-the-hsm-adoption.adoc new file mode 100644 index 000000000..df6275fe6 --- /dev/null +++ b/docs_user/modules/proc_rolling-back-the-hsm-adoption.adoc @@ -0,0 +1,40 @@ +:_mod-docs-content-type: PROCEDURE +[id="rolling-back-the-HSM-adoption_{context}"] + += Rolling back the HSM adoption + +[role="_abstract"] +If the hardware security module (HSM) adoption fails, you can restore your environment to its original state and attempt the adoption again. + +.Procedure + +. Restore the {rhos_long} 18.0 database backup: ++ +---- +$ oc exec -i openstack-galera-0 -n openstack -- mysql -u root -p barbican < /path/to/your/backups/rhoso18_barbican_backup.sql +---- ++ +where: + +:: +Specifies your database password. + +. Reset to standard images: ++ +---- +$ oc delete openstackversion openstack -n openstack +---- + +. Restore the base control plane configuration: ++ +---- +$ oc apply -f /path/to/your/base_controlplane.yaml +---- + +.Next steps + +To avoid additional issues when attempting your adoption again, consider the following suggestions: + +* Check the adoption logs that are stored in your configured working directory with timestamped summary reports. +* For HSM-specific issues, consult the Proteccio documentation and verify HSM connectivity from the target environment. +* Run the adoption in dry-run mode (`./run_proteccio_adoption.sh` option 3) to validate the environment before making changes. diff --git a/docs_user/modules/ref_troubleshooting-key-manager-hsm-adoption.adoc b/docs_user/modules/ref_troubleshooting-key-manager-hsm-adoption.adoc deleted file mode 100644 index 4f4c028eb..000000000 --- a/docs_user/modules/ref_troubleshooting-key-manager-hsm-adoption.adoc +++ /dev/null @@ -1,323 +0,0 @@ -:_mod-docs-content-type: ASSEMBLY -[id="troubleshooting-key-manager-hsm-adoption_{context}"] - -= Troubleshooting Key Manager HSM adoption - -:context: troubleshooting-hsm - - -[role="_abstract"] -Review troubleshooting guidance for common issues that you might encounter while you perform the HSM-enabled Key Manager (Barbican) service adoption. - -== Next steps -If issues persist after following the troubleshooting guide: - -* Collect adoption logs and configuration for analysis. -* Check the HSM vendor documentation for vendor-specific troubleshooting. -* Verify HSM server status and connectivity independently. -* Review the adoption summary report for additional diagnostic information. - -== Resolving configuration validation failures - -.Problem -[role="_abstract"] -If the adoption fails with validation errors about placeholder values, replace the placeholder values with your environment's configuration values. Validation errors about placeholder values look similar to the following example: -+ ----- -TASK [Validate all required variables are set] **** -fatal: [localhost]: FAILED! => { - "msg": "Required variable proteccio_certs_path contains placeholder value." -} ----- - - -.Procedure -. Edit your hardware security module configuration in the Zuul job vars or CI framework configuration file. - -. Replace all placeholder values with actual configuration values for your environment. Check the following key variables: -+ ----- -cifmw_hsm_password: -cifmw_barbican_proteccio_partition: -cifmw_barbican_proteccio_mkek_label: -cifmw_barbican_proteccio_hmac_label: -cifmw_hsm_proteccio_client_src: -cifmw_hsm_proteccio_conf_src: ----- - -. Verify that no placeholder values remain in your configuration. - -== Resolving missing HSM file prerequisites - -[role="_abstract"] -If the adoption fails because hardware security module (HSM) certificates or client software cannot be found, update your configuration to point to the files in their specific locations. The error looks similar to the following example: -+ ----- -TASK [Validate Proteccio prerequisites exist] **** -fatal: [localhost]: FAILED! => { - "msg": "Proteccio client ISO not found: /opt/proteccio/Proteccio3.06.05.iso" -} ----- - -.Procedure -. Verify that all required HSM files are accessible from the configured URLs: -+ -.Example -.Example -[source,bash] ----- -$ curl -I https://your-server/path/to/Proteccio3.06.05.iso -$ curl -I https://your-server/path/to/proteccio.rc -$ curl -I https://your-server/path/to/client.crt -$ curl -I https://your-server/path/to/client.key ----- - -. If the files are in different locations, update the URL variables in your configuration: -+ -.Example ----- -cifmw_hsm_proteccio_client_src: "https://correct-server/path/to/Proteccio3.06.05.iso" -cifmw_hsm_proteccio_conf_src: "https://correct-server/path/to/proteccio.rc" -cifmw_hsm_proteccio_client_crt_src: "https://correct-server/path/to/client.crt" -cifmw_hsm_proteccio_client_key_src: "https://correct-server/path/to/client.key" ----- - -. Check the network connectivity and authentication to ensure that the URLs are accessible from the CI environment. - -== Resolving source environment connectivity issues - -[role="_abstract"] -If the adoption cannot connect to the source {rhos_prev_long} environment to extract the configuration, check your SSH connectivity to the source Controller node and update the configuration if needed. The error for this issue looks similar to the following example: -+ ----- -TASK [detect source environment HSM configuration] **** -fatal: [localhost]: FAILED! => { - "msg": "SSH connection to source environment failed" -} ----- - -.Procedure -. Verify SSH connectivity to the source Controller node: -+ -[source,bash] ----- -$ ssh -o StrictHostKeyChecking=no tripleo-admin@controller-0.ctlplane ----- - -. Update the `controller1_ssh` variable if needed: -+ ----- -controller1_ssh: "ssh -o StrictHostKeyChecking=no tripleo-admin@" ----- - -. Ensure that the SSH keys are properly configured for passwordless access. - -== Resolving HSM secret creation failures - -.Problem -HSM secrets cannot be created in the target environment. - -.Symptoms ----- -TASK [Create HSM secrets in target environment] **** -fatal: [localhost]: FAILED! => { - "msg": "Failed to create secret proteccio-data" -} ----- - -[role="_abstract"] -If hardware security module (HSM) secrets cannot be created in the target environment, it might mean that you need to update the names of your secrets in your source configuration file. The error looks similar to the following example: -+ ----- -TASK [Create HSM secrets in target environment] **** -fatal: [localhost]: FAILED! => { - "msg": "Failed to create secret proteccio-data" -} ----- - -.Procedure -. Verify target environment access: -+ -[source,bash] ----- -$ export KUBECONFIG=/path/to/.kube/config -$ oc get secrets -n openstack ----- - -. Check if secrets already exist: -+ -[source,bash] ----- -$ oc get secret proteccio-data hsm-login -n openstack ----- - -. If secrets exist with different names, update the configuration variables: -+ -[source,yaml] ----- -proteccio_login_secret_name: "your-hsm-login-secret" -proteccio_client_data_secret_name: "your-proteccio-data-secret" ----- - -== Resolving custom image registry issues - -[role="_abstract"] -If you see the following error indicating that custom Barbican images cannot be pushed to or pulled from the configured registry, you can verify the authentication, test image push permissions, and then update the configuration as needed. -+ ----- -TASK [Create Proteccio-enabled Barbican images] **** -fatal: [localhost]: FAILED! => { - "msg": "Failed to push image to registry" -} ----- - -.Procedure -. Verify registry authentication: -+ -[source,bash] ----- -$ podman login ----- - -. Test image push permissions: -+ -[source,bash] ----- -$ podman tag hello-world //test:latest -$ podman push //test:latest ----- - -. Update registry configuration variables if needed: -+ -[source,yaml] ----- -cifmw_update_containers_registry: "your-registry:5001" -cifmw_update_containers_org: "your-namespace" -cifmw_image_registry_verify_tls: false ----- - -== Resolving HSM back-end detection failures - -[role="_abstract"] -If the adoption role cannot detect hardware security module (HSM) configuration in the source environment, you must force the HSM adoption. The error looks similar to the following example: -+ ----- -TASK [detect source environment HSM configuration] **** -ok: [localhost] => { - "msg": "No HSM configuration found - using standard adoption" -} ----- - -.Procedure -. Manually verify that the HSM configuration exists in the source environment: -+ -[source,bash] ----- -$ ssh tripleo-admin@controller-0.ctlplane \ - "sudo grep -A 10 '\[p11_crypto_plugin\]' \ - /var/lib/config-data/puppet-generated/barbican/etc/barbican/barbican.conf" ----- - -. If HSM is configured but not detected, force HSM adoption by setting the `barbican_hsm_enabled` variable: -+ -[source,yaml] ----- -# In your Zuul job vars or CI framework configuration -barbican_hsm_enabled: true ----- -+ -This configuration ensures that the `barbican_adoption` role uses the HSM-enabled patch for {key_manager_first_ref} deployment. - -== Resolving database migration issues - -[role="_abstract"] -If hardware security module (HSM) metadata is not preserved during database migration, check the database logs for any errors and verify that the source database includes the HSM secrets. The error looks similar to the following example: -+ ----- -TASK [Verify database migration preserves HSM references] **** -ok: [localhost] => { - "msg": "HSM secrets found in migrated database: 0" -} ----- - -.Procedure -. Verify that the source database contains the HSM secrets: -+ -[source,bash] ----- -$ ssh tripleo-admin@controller-0.ctlplane \ - "sudo mysql barbican -e 'SELECT COUNT(*) FROM secret_store_metadata WHERE key=\"plugin_name\" AND value=\"PKCS11\";'" ----- - -. Check the database migration logs for errors: -+ -[source,bash] ----- -$ oc logs deployment/barbican-api | grep -i migration ----- - -. If the migration failed, restore the database from backup and retry. - -== Resolving service startup failures - -[role="_abstract"] -If the {key_manager_first_ref} services fail to start after the hardware security module (HSM) configuration is applied, check the configuration in the pod. The error for the services failing to start looks similar to the following example: -+ ----- -$ oc get pods -l service=barbican -NAME READY STATUS RESTARTS AGE -barbican-api-xyz 0/1 Error 0 2m ----- - -.Procedure -. Check pod logs for HSM connectivity issues: -+ -[source,bash] ----- -$ oc logs barbican-api-xyz ----- - -. Verify HSM library is accessible: -+ -[source,bash] ----- -$ oc exec barbican-api-xyz -- ls -la /usr/lib64/libnethsm.so ----- - -. Check HSM configuration in the pod: -+ -[source,bash] ----- -$ oc exec barbican-api-xyz -- cat /etc/proteccio/proteccio.rc ----- - -== Resolving performance and connectivity issues - -[role="_abstract"] -If the hardware security module (HSM) operations are slow or fail intermittently, check the HSM connectivity and monitor the HSM server logs. - -.Procedure -. Test HSM connectivity from {key_manager_first_ref} pods: -+ -[source,bash] ----- -$ oc exec barbican-api-xyz -- pkcs11-tool --module /usr/lib64/libnethsm.so --list-slots ----- - -. Check HSM server connectivity: -+ -[source,bash] ----- -$ oc exec barbican-api-xyz -- nc -zv ----- - -. Monitor HSM server logs for authentication or capacity issues. - -== Getting additional help - -If issues persist after following this troubleshooting guide: - -. Collect adoption logs and configuration for analysis -. Check the HSM vendor documentation for vendor-specific troubleshooting -. Verify HSM server status and connectivity independently -. Review the adoption summary report for additional diagnostic information diff --git a/docs_user/modules/ref_troubleshooting-key-manager-proteccio-adoption.adoc b/docs_user/modules/ref_troubleshooting-key-manager-proteccio-adoption.adoc deleted file mode 100644 index 39fed6aab..000000000 --- a/docs_user/modules/ref_troubleshooting-key-manager-proteccio-adoption.adoc +++ /dev/null @@ -1,275 +0,0 @@ -:_mod-docs-content-type: ASSEMBLY -ifdef::context[:parent-context: {context}] -[id="troubleshooting-key-manager-proteccio-adoption_{context}"] -:context: troubleshooting-proteccio -= Troubleshooting {key_manager} Proteccio HSM adoption - -[role="_abstract"] -Use this reference to troubleshoot common issues that might occur during {key_manager_first_ref} adoption with Proteccio HSM integration. If Proteccio HSM issues persist, consult the Eviden Trustway documentation and ensure that HSM server configuration matches the client settings. - -== Resolving prerequisite validation failures - -[role="_abstract"] -The adoption script fails with the following errors during the prerequisites check, verify that your configuration includes all the required Proteccio files and that the HSM Ansible role is available. -+ ----- -ERROR: Required file proteccio_files/YOUR_CERT_FILE not found -ERROR: Cannot connect to OpenShift cluster -ERROR: Proteccio HSM Ansible role not found ----- - -.Procedure - -. Verify that all required Proteccio files are present: -+ ----- -$ ls -la /path/to/your/proteccio_files/ ----- -+ -Ensure that your configured certificate files, private key, HSM certificate file, and configuration file exist as specified in your `proteccio_required_files` configuration. - -. Test OpenShift cluster connectivity: -+ ----- -$ oc cluster-info -$ oc get pods -n openstack ----- - -. Verify that the HSM Ansible role is available: -+ ----- -$ ls -la /path/to/your/roles/ansible-role-rhoso-proteccio-hsm/ ----- - -== Resolving SSH connection failures to the source environment - -[role="_abstract"] -If you cannot connect to source {OpenStackPreviousInstaller} environment, you might see the following error: -+ ----- -Warning: Permanently added 'YOUR_UNDERCLOUD_HOST' (ED25519) to the list of known hosts. -Permission denied (publickey). ----- - -To troubleshoot the issue, verify your SSH key access and test the SSH commands that the adoption uses. - -*Solution*: - -. Verify SSH key access to the undercloud: -+ ----- -$ ssh YOUR_UNDERCLOUD_HOST echo "Connection test" ----- - -. Test the specific SSH commands used by the adoption: -+ ----- -$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack bash -lc "echo test"' -$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "echo test"' ----- - -. If the connection fails, verify the SSH configuration and ensure that the undercloud hostname resolves correctly. - -== Resolving database import failures - -[role="_abstract"] -If the source database export or import fails, check the the source Galera container, database connectivity, and the source {key_manager_first_ref} configuration. The database import or export error looks similar to the following example: -+ ----- -Error: no container with name or ID "galera-bundle-podman-0" found -mysqldump: Got error: 1045: "Access denied for user 'barbican'@'localhost'" ----- - -.Procedure - -. Verify that the source Galera container is running: -+ ----- -$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo podman ps | grep galera"' ----- - -. Test database connectivity with the extracted credentials: -+ ----- -$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo podman exec galera-bundle-podman-0 mysql -u barbican -p -e \"SELECT 1;\""' ----- - -. Check the source {key_manager} configuration for the correct database password: -+ ----- -$ sudo ssh -t YOUR_UNDERCLOUD_HOST 'sudo -u stack ssh -t tripleo-admin@YOUR_CONTROLLER_HOST.ctlplane "sudo grep connection /var/lib/config-data/puppet-generated/barbican/etc/barbican/barbican.conf"' ----- - -== Resolving custom image pull failures - -[role="_abstract"] -If Proteccio custom images fail to pull or start, you might see the following error: -+ ----- -Failed to pull image "": rpc error -Pod has unbound immediate PersistentVolumeClaims ----- - -To troubleshoot this issue, verify image registry access, image pull secrets, and registry authentication. - -.Procedure - -. Verify image registry access: -+ ----- -$ podman pull ----- - -. Check image pull secrets and registry authentication: -+ ----- -$ oc get secrets -n openstack | grep pull -$ oc describe pod -n openstack ----- - -. Verify that the `OpenStackVersion` resource was applied correctly: -+ ----- -$ oc get openstackversion openstack -n openstack -o yaml ----- - -== Resolving HSM certificate mounting issues - -[role="_abstract"] -If Proteccio client certificates are not properly mounted in pods, check the secret creation and ensure that the {key_manager_first_ref} configuration includes the correct volume mounts. The error looks similar to the following example: -+ ----- -$ oc exec -c barbican-api -- ls -la /etc/proteccio/ -ls: cannot access '/etc/proteccio/': No such file or directory ----- - -.Procedure - -. Verify that the `proteccio-data` secret was created correctly: -+ ----- -$ oc describe secret proteccio-data -n openstack ----- - -. Check that the secret contains the expected files: -+ ----- -$ oc get secret proteccio-data -n openstack -o yaml ----- - -. Verify that the {key_manager} configuration includes the correct volume mounts: -+ ----- -$ oc get barbican barbican -n openstack -o yaml | grep -A10 pkcs11 ----- - -== Resolving service startup failures - -[role="_abstract"] -If {key_manager_first_ref} services fail to start after configuration, check the pod logs, RabbitMQ user configuration, and resource constraints. The error looks similar to the following example: -+ ----- -CrashLoopBackOff -Init:Error -amqp.exceptions.AccessRefused: Login was refused using authentication mechanism AMQPLAIN ----- - -.Procedure - -. Check pod logs for specific error messages: -+ ----- -$ oc logs -c barbican-api -n openstack -$ oc logs -c barbican-api-log -n openstack ----- - -. Verify that the {key_manager} configuration is valid: -+ ----- -$ oc get barbican barbican -n openstack -o yaml ----- - -. Check the RabbitMQ user configuration if you see authentication errors: -+ ----- -# Get the transport URL to find expected username -$ oc get secret rabbitmq-transport-url-barbican-barbican-transport -n openstack -o jsonpath='{.data.transport_url}' | base64 -d - -# Create the missing RabbitMQ user (extract username and password from URL above) -$ oc exec rabbitmq-server-0 -n openstack -- rabbitmqctl add_user -$ oc exec rabbitmq-server-0 -n openstack -- rabbitmqctl set_permissions ".*" ".*" ".*" - -# Restart failing pods -$ oc delete pods -l service=barbican -n openstack ----- - -. Check for resource constraints or scheduling issues: -+ ----- -$ oc describe pod -n openstack ----- - -== Resolving adoption verification failures - -[role="_abstract"] -If the secrets from the source environment are not accessible after adoption, you might see the following error: -+ ----- -$ openstack secret list -# Returns empty list or HTTP 500 errors ----- - -To troubleshoot this issue, verify that the database import completed successfully, test API connectivity, and check for schema adoption issues. - -.Procedure - -. Verify that the database import completed successfully: -+ ----- -$ oc exec openstack-galera-0 -n openstack -- mysql -u root -p barbican -e "SELECT COUNT(*) FROM secrets;" ----- - -. Check for schema adoption issues: -+ ----- -$ oc logs job.batch/barbican-db-sync -n openstack ----- - -. Test API connectivity: -+ ----- -$ oc exec openstackclient -n openstack -- curl -s -k -H "X-Auth-Token: $(openstack token issue -f value -c id)" https://barbican-internal.openstack.svc:9311/v1/secrets ----- - -. Verify that projects and users were adopted correctly, as secrets are project-scoped. - -== Rolling back the HSM adoption - -[role="_abstract"] -If the hardware security module (HSM) adoption fails, you can restore your environment to its original state and attempt the adoption again. - -. Restore the {rhos_long} 18.0 database backup: -+ ----- -$ oc exec -i openstack-galera-0 -n openstack -- mysql -u root -p barbican < /path/to/your/backups/rhoso18_barbican_backup.sql ----- - -. Reset to standard images: -+ ----- -$ oc delete openstackversion openstack -n openstack ----- - -. Restore the base control plane configuration: -+ ----- -$ oc apply -f /path/to/your/base_controlplane.yaml ----- - -.Next steps - -To avoid additional issues when attempting your adoption again, consider the following suggestions: - -* Check the adoption logs that are stored in your configured working directory with timestamped summary reports. -* For HSM-specific issues, consult the Proteccio documentation and verify HSM connectivity from the target environment. -* Run the adoption in dry-run mode (`./run_proteccio_adoption.sh` option 3) to validate the environment before making changes. diff --git a/scenarios/hci.yaml b/scenarios/hci.yaml index 8d3afa51f..7fea1cfc1 100644 --- a/scenarios/hci.yaml +++ b/scenarios/hci.yaml @@ -3,7 +3,7 @@ undercloud: config: - section: DEFAULT option: undercloud_hostname - value: undercloud.example.com + value: osp-undercloud-0.ooo.test - section: DEFAULT option: undercloud_timezone value: UTC @@ -22,7 +22,8 @@ undercloud: undercloud_parameters_override: "hci/hieradata_overrides_undercloud.yaml" undercloud_parameters_defaults: "hci/undercloud_parameter_defaults.yaml" ctlplane_vip: 192.168.122.98 -cloud_domain: "example.com" +cloud_domain: "ooo.test" +tlse: true hostname_groups_map: # map ansible groups in the inventory to role hostname format for # 17.1 deployment @@ -58,6 +59,9 @@ stacks: - "/usr/share/openstack-tripleo-heat-templates/environments/cephadm/ceph-dashboard.yaml" - "/usr/share/openstack-tripleo-heat-templates/environments/services/barbican.yaml" - "/usr/share/openstack-tripleo-heat-templates/environments/barbican-backend-simple-crypto.yaml" + - "/usr/share/openstack-tripleo-heat-templates/environments/ssl/tls-everywhere-endpoints-dns.yaml" + - "/usr/share/openstack-tripleo-heat-templates/environments/services/haproxy-public-tls-certmonger.yaml" + - "/usr/share/openstack-tripleo-heat-templates/environments/ssl/enable-internal-tls.yaml" network_data_file: "hci/network_data.yaml.j2" vips_data_file: "hci/vips_data.yaml" roles_file: "hci/roles.yaml" diff --git a/scenarios/hci/config_download.yaml b/scenarios/hci/config_download.yaml index 39fd67e0b..9a3df4696 100644 --- a/scenarios/hci/config_download.yaml +++ b/scenarios/hci/config_download.yaml @@ -34,6 +34,7 @@ resource_registry: OS::TripleO::Services::HeatApiCfn: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-api-cfn-container-puppet.yaml OS::TripleO::Services::HeatApiCloudwatch: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-api-cloudwatch-disabled-puppet.yaml OS::TripleO::Services::HeatEngine: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-engine-container-puppet.yaml + OS::TripleO::Services::IpaClient: /usr/share/openstack-tripleo-heat-templates/deployment/ipa/ipaservices-baremetal-ansible.yaml parameter_defaults: RedisVirtualFixedIPs: - ip_address: 192.168.122.110 @@ -55,18 +56,21 @@ parameter_defaults: ComputeHCICount: 3 NeutronGlobalPhysnetMtu: 1350 CinderLVMLoopDeviceSize: 20480 - CloudName: overcloud.example.com - CloudNameInternal: overcloud.internalapi.example.com - CloudNameStorage: overcloud.storage.example.com - CloudNameStorageManagement: overcloud.storagemgmt.example.com - CloudNameCtlplane: overcloud.ctlplane.example.com - CloudDomain: example.com + CloudName: overcloud.ooo.test + CloudNameInternal: overcloud.internalapi.ooo.test + CloudNameStorage: overcloud.storage.ooo.test + CloudNameStorageManagement: overcloud.storagemgmt.ooo.test + CloudNameCtlplane: overcloud.ctlplane.ooo.test + CloudDomain: ooo.test + IdMServer: osp-free-ipa-0.ooo.test + IdMDomain: ooo.test + IdMInstallClientPackages: true NetworkConfigWithAnsible: false ControllerHostnameFormat: '%stackname%-controller-%index%' ComputeHCIHostnameFormat: '%stackname%-computehci-%index%' CtlplaneNetworkAttributes: network: - dns_domain: example.com + dns_domain: ooo.test mtu: 1500 name: ctlplane tags: diff --git a/scenarios/uni01alpha/config_download.yaml b/scenarios/uni01alpha/config_download.yaml index 445333dd0..94e5b4b95 100644 --- a/scenarios/uni01alpha/config_download.yaml +++ b/scenarios/uni01alpha/config_download.yaml @@ -35,6 +35,11 @@ resource_registry: OS::TripleO::Services::HeatApiCfn: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-api-cfn-container-puppet.yaml OS::TripleO::Services::HeatApiCloudwatch: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-api-cloudwatch-disabled-puppet.yaml OS::TripleO::Services::HeatEngine: /usr/share/openstack-tripleo-heat-templates/deployment/heat/heat-engine-container-puppet.yaml + OS::TripleO::Services::OctaviaApi: /usr/share/openstack-tripleo-heat-templates//deployment/octavia/octavia-api-container-puppet.yaml + OS::TripleO::Services::OctaviaHousekeeping: /usr/share/openstack-tripleo-heat-templates/deployment/octavia/octavia-housekeeping-container-puppet.yaml + OS::TripleO::Services::OctaviaHealthManager: /usr/share/openstack-tripleo-heat-templates/deployment/octavia/octavia-health-manager-container-puppet.yaml + OS::TripleO::Services::OctaviaWorker: /usr/share/openstack-tripleo-heat-templates/deployment/octavia/octavia-worker-container-puppet.yaml + OS::TripleO::Services::OctaviaDeploymentConfig: /usr/share/openstack-tripleo-heat-templates/deployment/octavia/octavia-deployment-config.yaml parameter_defaults: RedisVirtualFixedIPs: - ip_address: 192.168.122.130 @@ -48,6 +53,7 @@ parameter_defaults: ComputeExtraConfig: nova::compute::libvirt::services::libvirt_virt_type: qemu nova::compute::libvirt::virt_type: qemu + NeutronDnsDomain: 'openstackgate.local' BarbicanSimpleCryptoGlobalDefault: true Debug: true DockerPuppetDebug: true @@ -55,7 +61,7 @@ parameter_defaults: ControllerCount: 3 ComputeCount: 2 NetworkerCount: 3 - NeutronGlobalPhysnetMtu: 1350 + NeutronGlobalPhysnetMtu: 1500 CinderLVMLoopDeviceSize: 20480 CloudName: overcloud.localdomain CloudNameInternal: overcloud.internalapi.localdomain @@ -111,3 +117,9 @@ parameter_defaults: IronicNetwork: ironic IronicInspectorNetwork: ironic IronicCleaningDiskErase: metadata + NeutronEnableForceMetadata: true + # This flag enables internal generation of certificates for communication + # with amphorae. Use OctaviaCaCert, OctaviaCaKey, OctaviaCaKeyPassphrase, + # OctaviaClient and OctaviaServerCertsKeyPassphrase cert to configure + # secure production environments. + OctaviaGenerateCerts: true diff --git a/scenarios/uni07eta/config_download.yaml b/scenarios/uni07eta/config_download.yaml index c2fa4274a..ad5d56eaf 100644 --- a/scenarios/uni07eta/config_download.yaml +++ b/scenarios/uni07eta/config_download.yaml @@ -11,6 +11,9 @@ resource_registry: OS::TripleO::Controller::Ports::InternalApiPort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_internal_api.yaml OS::TripleO::Controller::Ports::StoragePort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_storage.yaml OS::TripleO::Controller::Ports::TenantPort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_tenant.yaml + OS::TripleO::Networker::Ports::InternalApiPort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_internal_api.yaml + OS::TripleO::Networker::Ports::StoragePort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_storage.yaml + OS::TripleO::Networker::Ports::TenantPort: /usr/share/openstack-tripleo-heat-templates/network/ports/deployed_tenant.yaml OS::TripleO::Services::CeilometerAgentCentral: /usr/share/openstack-tripleo-heat-templates/deployment/ceilometer/ceilometer-agent-central-container-puppet.yaml OS::TripleO::Services::CeilometerAgentNotification: /usr/share/openstack-tripleo-heat-templates/deployment/ceilometer/ceilometer-agent-notification-container-puppet.yaml OS::TripleO::Services::CeilometerAgentIpmi: /usr/share/openstack-tripleo-heat-templates/deployment/ceilometer/ceilometer-agent-ipmi-container-puppet.yaml @@ -45,6 +48,10 @@ parameter_defaults: ComputeExtraConfig: nova::compute::libvirt::services::libvirt_virt_type: qemu nova::compute::libvirt::virt_type: qemu + ExtraConfig: + neutron::notification_driver: 'noop' + neutron::plugins::ml2::path_mtu: 1500 + neutron::plugins::ml2::ovn::ovn_router_indirect_snat: true BarbicanSimpleCryptoGlobalDefault: true Debug: true DockerPuppetDebug: true @@ -52,7 +59,7 @@ parameter_defaults: ControllerCount: 3 ComputeCount: 2 NetworkerCount: 3 - NeutronGlobalPhysnetMtu: 1350 + NeutronGlobalPhysnetMtu: 1500 CinderLVMLoopDeviceSize: 20480 CloudName: overcloud.example.com CloudNameInternal: overcloud.internalapi.example.com @@ -61,6 +68,14 @@ parameter_defaults: CloudNameCtlplane: overcloud.ctlplane.example.com CloudDomain: example.com NetworkConfigWithAnsible: false + EnableVLANTransparency: true + NeutronEnableIgmpSnooping: true + OVNEmitNeedToFrag: true + NeutronEnableDVR: true + NeutronTypeDrivers: 'geneve,vxlan,vlan,flat,local' + NeutronNetworkType: 'geneve,flat,vlan' + NeutronDnsDomain: 'example.com' + NeutronRouterSchedulerDriver: 'neutron.scheduler.l3_agent_scheduler.ChanceScheduler' ControllerHostnameFormat: '%stackname%-controller-%index%' ComputeHostnameFormat: '%stackname%-compute-%index%' NetworkerHostnameFormat: '%stackname%-networker-%index%' diff --git a/scenarios/uni07eta/network_data.yaml.j2 b/scenarios/uni07eta/network_data.yaml.j2 index c582126ea..675aa1df3 100644 --- a/scenarios/uni07eta/network_data.yaml.j2 +++ b/scenarios/uni07eta/network_data.yaml.j2 @@ -41,7 +41,7 @@ name_lower: octavia dns_domain: octavia.{{ cloud_domain }}. subnets: - octavie_subnet: + octavia_subnet: ip_subnet: 172.23.0.0/24 allocation_pools: - start: 172.23.0.200 @@ -55,6 +55,6 @@ service_net_map_replace: external subnets: external_subnet: - vlan: 44 - ip_subnet: '10.0.0.0/24' - allocation_pools: [{'start': '10.0.0.150', 'end': '10.0.0.250'}] + vlan: 218 + ip_subnet: '172.38.0.0/24' + allocation_pools: [{'start': '172.38.0.50', 'end': '172.38.0.80'}] diff --git a/scenarios/uni07eta/roles.yaml b/scenarios/uni07eta/roles.yaml index d58941878..eb7d9c7b7 100644 --- a/scenarios/uni07eta/roles.yaml +++ b/scenarios/uni07eta/roles.yaml @@ -260,6 +260,8 @@ subnet: internal_api_subnet Tenant: subnet: tenant_subnet + External: + subnet: external_subnet tags: - external_bridge RoleParametersDefault: diff --git a/scenarios/uni07eta/undercloud_parameter_defaults.yaml b/scenarios/uni07eta/undercloud_parameter_defaults.yaml index 64e2481da..55bb408d9 100644 --- a/scenarios/uni07eta/undercloud_parameter_defaults.yaml +++ b/scenarios/uni07eta/undercloud_parameter_defaults.yaml @@ -1,14 +1,5 @@ --- { - "parameter_defaults": { - "MasqueradeNetworks": { - "10.0.0.1/24": [ - "10.0.0.1/24" - ], - "192.168.122.0/24": [ - "192.168.122.0/24" - ] - } - }, + "parameter_defaults": {}, "resource_registry": {} } diff --git a/scenarios/uni07eta/vips_data.yaml b/scenarios/uni07eta/vips_data.yaml index e7541dde0..664a01f9b 100644 --- a/scenarios/uni07eta/vips_data.yaml +++ b/scenarios/uni07eta/vips_data.yaml @@ -13,6 +13,6 @@ dns_name: overcloud - name: ctlplane_vip network: ctlplane - ip_address: 192.168.122.101 + ip_address: 192.168.122.99 subnet: ctlplane-subnet dns_name: overcloud diff --git a/tests/roles/common_defaults/defaults/main.yaml b/tests/roles/common_defaults/defaults/main.yaml index 351675bfd..241bca852 100644 --- a/tests/roles/common_defaults/defaults/main.yaml +++ b/tests/roles/common_defaults/defaults/main.yaml @@ -224,6 +224,10 @@ pull_openstack_configuration_ssh_shell_vars: | enable_octavia: true octavia_adoption: true +# Whether to create Octavia workload before adoption - defined in common_defaults because it is used in two roles: +# dataplane_adoption and development_environment +prelaunch_octavia_workload: false + # MariaDB client connection timeout in seconds # Related to OSPRH-18618 mariadb_client_timeout: 0 diff --git a/tests/roles/dataplane_adoption/defaults/main.yaml b/tests/roles/dataplane_adoption/defaults/main.yaml index 611ad833a..8726a52c1 100644 --- a/tests/roles/dataplane_adoption/defaults/main.yaml +++ b/tests/roles/dataplane_adoption/defaults/main.yaml @@ -204,9 +204,6 @@ edpm_network_config_template_bgp: | addresses: - ip_netmask: {{ lookup('vars', 'bgpmainnet_ip') }}/32 - ip_netmask: {{ lookup('vars', 'bgpmainnetv6_ip') }}/128 - - ip_netmask: {{ lookup('vars', 'internalapi_ip') }}/32 - - ip_netmask: {{ lookup('vars', 'storage_ip') }}/32 - - ip_netmask: {{ lookup('vars', 'tenant_ip') }}/32 {% endraw %} neutron_physical_bridge_name: br-ctlplane neutron_public_interface_name: "{{ dataplane_public_iface | default('eth0') }}" @@ -232,7 +229,6 @@ skip_patching_ansibleee_csv: false os_diff_dir: tmp/os-diff os_diff_data_dir: tmp/os-diff prelaunch_test_instance: true -prelaunch_octavia_workload: false telemetry_adoption: true # nodes data will be templated in as a separate @@ -374,7 +370,7 @@ dataplane_cr: | {% for login, value in edpm_container_registry_logins.items() -%} {{ login }}: {{ value | to_nice_yaml | trim }} - {%- endfor %} + {% endfor %} {%- endif %} gather_facts: false @@ -556,6 +552,14 @@ networker_cr: | edpm_ovn_ofctrl_wait_before_clear: 8000 edpm_enable_chassis_gw: true + {% if edpm_container_registry_logins is defined -%} + edpm_container_registry_logins: + {% for login, value in edpm_container_registry_logins.items() -%} + {{ login }}: + {{ value | to_nice_yaml | trim }} + {% endfor %} + {%- endif %} + {% if edpm_container_registry_insecure_registries is defined -%} edpm_container_registry_insecure_registries: {{ edpm_container_registry_insecure_registries }} {% endif -%} diff --git a/tests/roles/dataplane_adoption/tasks/main.yaml b/tests/roles/dataplane_adoption/tasks/main.yaml index 512be18cc..ddc5dece7 100644 --- a/tests/roles/dataplane_adoption/tasks/main.yaml +++ b/tests/roles/dataplane_adoption/tasks/main.yaml @@ -647,6 +647,22 @@ done when: edpm_neutron_dhcp_agent_enabled|bool +- name: enable neutron-dhcp in the OpenStackDataPlaneNodeSet CR Networker + no_log: "{{ use_no_log }}" + ansible.builtin.shell: | + {{ shell_header }} + {{ oc_header }} + + oc patch openstackdataplanenodeset openstack-networker --type='json' --patch='[ + { + "op": "add", + "path": "/spec/services/-", + "value": "neutron-dhcp" + }]' + when: + - edpm_networker_neutron_dhcp_agent_enabled | default(false) | bool + - edpm_nodes_networker is defined + - name: Run the pre-adoption validation when: run_pre_adoption_validation|bool block: diff --git a/zuul.d/jobs-layout.yaml b/zuul.d/jobs-layout.yaml index 0bc77ef93..5fd53b656 100644 --- a/zuul.d/jobs-layout.yaml +++ b/zuul.d/jobs-layout.yaml @@ -4,6 +4,9 @@ github-check: jobs: - noop - - adoption-standalone-to-crc-ceph - - adoption-standalone-to-crc-no-ceph - - adoption-docs-preview + - adoption-standalone-to-crc-ceph: &required_projects + required-projects: + - name: openstack-k8s-operators/data-plane-adoption + override-checkout: 18.0-fr5 + - adoption-standalone-to-crc-no-ceph: *required_projects + - adoption-docs-preview: *required_projects