Loading src/opticalcontroller/Dockerfile +1 −1 Changes for src/opticalcontroller/Dockerfile: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -20,7 +20,7 @@ RUN apt-get --yes --quiet --quiet update && \ rm -rf /var/lib/apt/lists/* # Set Python to show logs as they occur ENV PYTHONUNBUFFERED=0 ENV PYTHONUNBUFFERED=1 # Download the gRPC health probe RUN GRPC_HEALTH_PROBE_VERSION=v0.2.0 && \ Loading src/opticalcontroller/OpticalController.py +165 −9 Changes for src/opticalcontroller/OpticalController.py: 165 added lines, 9 removed lines. Original line number Diff line number Diff line Loading @@ -45,7 +45,6 @@ from google.protobuf.json_format import MessageToDict from common.tools.object_factory.OpticalLink import order_dict from opticalcontroller.RSA import RSA from opticalcontroller.RSA_P2MP import RSA_P2MP from opticalcontroller.dscm_monitoring import get_status_snapshot from common.tools.object_factory.Context import json_context_id from common.tools.object_factory.Topology import json_topology_id Loading Loading @@ -1222,8 +1221,21 @@ def process_topology(context_id: str, topology_id: str): OPTICAL_ROADM_TYPES = { DeviceTypeEnum.OPTICAL_ROADM.value, DeviceTypeEnum.EMULATED_OPTICAL_ROADM.value } # Includes PACKET_ROUTER alongside the real optical-transponder types: # ensure_topology.py deliberately registers the T1.1/T1.2/T2.1 # pluggables with device_type='packet-router' (not their descriptor's # real 'optical-transponder') so TelemetryBackendService's collector # filter -- which always matches on device_type=PACKET_ROUTER # regardless of the target device -- can find them. Without this, # those devices were silently excluded from node_dict below, so they # were never vertices in the RSA path-computation graph at all: every # link touching them got filtered out a few lines down, and # ComputeP2MP/PathRecomp always fell through to "No valid forward # path found" -- a hollow 201 with no subcarrier-attributes -- no # matter how the topology's links were configured. OPTICAL_TRANSPONDER_TYPES = { DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value, DeviceTypeEnum.PACKET_ROUTER.value, DeviceTypeEnum.EMULATED_PACKET_ROUTER.value, } added_device_uuids = set() Loading Loading @@ -1869,6 +1881,7 @@ class ComputeP2MP(Resource): _register_path_computation( sources, destinations, request_data, flow_id, via="ComputeP2MP" ) _maybe_revert_on_source_recompute(sources) LOGGER.info("******************************************") LOGGER.info("TAPI Path Computation successfully retrieved") LOGGER.info(f"TAPI Path Computation Response: {tapi_response}") Loading Loading @@ -2324,6 +2337,7 @@ class PathRecomp(ComputeP2MP): _register_path_computation( sources, destinations, request_data, flow_id, via="pathrecomp" ) _maybe_revert_on_source_recompute(sources) LOGGER.info("PathRecomp: new path registered for destinations=%s flow_id=%s", destinations, flow_id) Loading Loading @@ -2920,6 +2934,73 @@ def _start_leaf_monitoring(device_name: str) -> None: automation_client.close() # ── T1.2 mute/confirm/revert state machine ───────────────────────────────── # Hardcoded specifically for the T1.1<->T1.2 pair (not a generic per-device # mechanism -- the existing _confirmed_up_devices gate above already plays # that role for whichever device _LEAF_DEVICE_NAME currently is, and is left # untouched here). # # T1.2's own resting/quiet RECEIVED_POWER band -- distinct from T1.1's # _REARM_RECEIVED_POWER_RANGE_DBM above since it's a different physical # channel with its own baseline. Right after failover, T1.2 sitting in this # band with PRE_FEC_BER exactly 0.0 doesn't yet prove it's confirmed up -- # just that it hasn't started reporting real measurements yet -- so its # alarms are muted until a genuine (non-zero) in-range PRE_FEC_BER is seen # together with an in-range RECEIVED_POWER. Once confirmed, settling back to # PRE_FEC_BER=0.0 with power back in this same resting band is read as "the # T1.2 path is quiet again" and reverts monitoring back to T1.1. _T1_2_RESTING_POWER_RANGE_DBM = (-6.0, -3.0) # True once T1.2 has actually been confirmed up (see # _update_t1_2_confirmation_state) since the most recent failover to it -- # its alarms are muted until then. Reset to False every time a fresh # failover to T1.2 happens or monitoring reverts back to T1.1. _t1_2_confirmed_up = False def _update_t1_2_confirmation_state(device_name: str, kpi_name_upper: str, measured_value) -> None: """Run for every RECEIVED_POWER/PRE_FEC_BER sample PostAlarmNotification processes, but only actually does anything once T1.2 is the currently monitored device (i.e. a failover has already happened). Drives the mute -> confirm -> revert cycle described above; the caller in PostAlarmNotification is responsible for actually suppressing T1.2's alarms while not yet confirmed (see _t1_2_confirmed_up there).""" global _t1_2_confirmed_up if device_name != _LEAF_FAILOVER_DEVICE_NAME or _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME: return if "RECEIVED_POWER" not in kpi_name_upper and "PRE_FEC_BER" not in kpi_name_upper: return power = _last_received_power_by_device.get(device_name, (None, None))[0] ber = _last_pre_fec_ber_by_device.get(device_name) if power is None or ber is None: return # need both readings at least once before this can decide anything resting = _T1_2_RESTING_POWER_RANGE_DBM[0] <= power <= _T1_2_RESTING_POWER_RANGE_DBM[1] if not _t1_2_confirmed_up: if ber != 0.0 and PRE_FEC_BER_RANGE[0] <= ber <= PRE_FEC_BER_RANGE[1] \ and RECEIVED_POWER_RANGE_DBM[0] <= power <= RECEIVED_POWER_RANGE_DBM[1]: _t1_2_confirmed_up = True LOGGER.info( "T1.2 confirmed up: genuine in-range PRE_FEC_BER=%s with " "RECEIVED_POWER=%s -- alarm monitoring now trusted for T1.2.", ber, power, ) else: LOGGER.debug( "T1.2 not yet confirmed up (PRE_FEC_BER=%s, RECEIVED_POWER=%s) " "-- still muted.", ber, power, ) elif ber == 0.0 and resting: LOGGER.info( "T1.2 settled back to its resting baseline (PRE_FEC_BER=0.0, " "RECEIVED_POWER=%s in %s) -- reverting monitoring back to %s.", power, _T1_2_RESTING_POWER_RANGE_DBM, _ORIGINAL_LEAF_DEVICE_NAME, ) _revert_to_original_leaf_device() def _switch_monitored_device_and_recompute(notification: dict) -> None: """Called right after the first genuinely new alarm is actually delivered to a subscriber (see the two call sites below) -- replaces the old Loading @@ -2941,7 +3022,7 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None: Runs at most once: _leaf_failover_done latches after the switch, so the device's later alarms -- now resolved against the new monitored device -- don't re-trigger it, and the VOA is never pushed to again.""" global _LEAF_DEVICE_NAME, _leaf_failover_done global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up if _leaf_failover_done: return Loading Loading @@ -3013,6 +3094,44 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None: _LEAF_DEVICE_NAME = _LEAF_FAILOVER_DEVICE_NAME _leaf_failover_done = True _t1_2_confirmed_up = False def _revert_to_original_leaf_device() -> None: """Actually perform the revert from the failover device back to _ORIGINAL_LEAF_DEVICE_NAME: restart real monitoring on it and reset the failover latch/T1.2-confirmation state so a future alarm can trigger the same failover again from a clean baseline. Shared by _restore_monitored_device_after_delete (path-delete triggered), _update_t1_2_confirmation_state (T1.2 quiet-baseline triggered), and _maybe_revert_on_source_recompute (path-computation-sourced-from-T1.1 triggered).""" global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME) _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME _leaf_failover_done = False _t1_2_confirmed_up = False def _maybe_revert_on_source_recompute(sources) -> None: """If a path computation/recompute request names _ORIGINAL_LEAF_DEVICE_NAME (T1.1) as one of its sources while monitoring is currently on the failover device, that's a clear, explicit signal that T1.1 is back in service -- revert monitoring back to it immediately, the same as the path-delete and T1.2-quiet-baseline triggers do. Called from ComputeP2MP and PathRecomp (which also covers /pathrecomp and /recompute-p2mp) right after either one successfully registers a new path.""" if _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME: return # not currently failed over -- nothing to revert if _ORIGINAL_LEAF_DEVICE_NAME not in (sources or []): return LOGGER.info( "Path computation sourced from %s while monitoring %s -- " "reverting monitoring back to %s.", _ORIGINAL_LEAF_DEVICE_NAME, _LEAF_FAILOVER_DEVICE_NAME, _ORIGINAL_LEAF_DEVICE_NAME, ) _revert_to_original_leaf_device() def _restore_monitored_device_after_delete(deleted: dict) -> None: Loading Loading @@ -3043,9 +3162,7 @@ def _restore_monitored_device_after_delete(deleted: dict) -> None: "restoring monitoring to %s.", failed_over_device, _ORIGINAL_LEAF_DEVICE_NAME, ) _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME) _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME _leaf_failover_done = False _revert_to_original_leaf_device() def _push_notification_to_webhook(notification: dict) -> None: Loading @@ -3061,7 +3178,7 @@ def _push_notification_to_webhook(notification: dict) -> None: try: # The receiver validates against the full TAPI notification-service # schema, which wraps each notification in a "notification-context" # array (see tests/ecoc26_pluggables/tests/test_e2e.py's # array (see tests/ecoc26_pluggables/tests/.py's # DEFAULT_PAYLOAD) -- posting the bare "...notification" object by # itself is rejected with a 400 "required property" error. resp = requests.post( Loading Loading @@ -3279,13 +3396,30 @@ class PostAlarmNotification(Resource): # Baseline-health gate (see _record_breach_state_and_check_confirmed_up # above): suppress every alarm for this device until PRE_FEC_BER and # RECEIVED_POWER have both been observed in range at once. # # The gate is fed gate_low_breached/gate_high_breached rather than # the raw low_breached/high_breached: a raw PRE_FEC_BER==0.0 fails # its own range check (0.0 is below PRE_FEC_BER_RANGE's floor), but # 0.0 is the best possible BER reading on its own (see # is_pre_fec_ber_zero above) -- treating it as a breach here meant a # device whose healthy baseline is a clean 0.0 could NEVER satisfy # the gate, leaving it (and every real fault on it afterward) muted # forever. This correction only affects what the gate records -- # the actual alarm decision below (including the RECEIVED_POWER # correlation that can still escalate a 0.0 reading to a real # failure) keeps using the original low_breached/high_breached, # unaffected by this. gate_low_breached, gate_high_breached = low_breached, high_breached if is_pre_fec_ber_zero: gate_low_breached, gate_high_breached = False, False metric = ("PRE_FEC_BER" if "PRE_FEC_BER" in kpi_name_upper else "RECEIVED_POWER" if "RECEIVED_POWER" in kpi_name_upper else None) confirmed_up = True if metric is not None: confirmed_up = _record_breach_state_and_check_confirmed_up( device_name, metric, low_breached, high_breached) device_name, metric, gate_low_breached, gate_high_breached) if not confirmed_up: # Silent while unconfirmed -- this fires on every single # sample until the device is confirmed up, so it stays at Loading Loading @@ -3335,6 +3469,28 @@ class PostAlarmNotification(Resource): ) low_breached = False # T1.2-specific mute/confirm/revert state machine (see # _update_t1_2_confirmation_state above) -- runs independently of, # and after, the generic confirmed-up gate above. Only actually # takes effect once T1.2 is the currently monitored device (i.e. # after a failover); a no-op otherwise. Has the final say for T1.2's # own alarms: even if the checks above decided this is a genuine # breach, T1.2 stays muted until it's been confirmed up in its own # right. if metric is not None and device_name: _update_t1_2_confirmation_state(device_name, kpi_name_upper, measured_value) if (device_name == _LEAF_FAILOVER_DEVICE_NAME and _LEAF_DEVICE_NAME == _LEAF_FAILOVER_DEVICE_NAME and not _t1_2_confirmed_up): LOGGER.debug( "T1.2 not yet confirmed up -- suppressing alarm for service %s / " "KPI %s (value=%s).", data.get("service_id", ""), kpi_context["kpi-name"], measured_value, ) g.suppress_latency_log = True low_breached = False high_breached = False if not low_breached and not high_breached: # DEBUG, not INFO: this fires on every single non-breaching # sample (i.e. most of them, most of the time) -- only a Loading Loading @@ -3380,7 +3536,7 @@ class PostAlarmNotification(Resource): "probable-cause": "THRESHOLD_CROSSED", # The receiver's TAPI schema requires these as numbers, not # booleans (validated against tests/ecoc26_pluggables/tests/ # test_e2e.py's DEFAULT_PAYLOAD) -- policyservice only tells # .py's DEFAULT_PAYLOAD) -- policyservice only tells # us whether each threshold was breached, not the KPI's # actual configured threshold value, so 1.0/0.0 preserves # that breach flag as a number instead of fabricating a Loading src/tests/PMPv1/descriptors/singleEndpoint.json +2 −2 Changes for src/tests/PMPv1/descriptors/singleEndpoint.json: 2 added lines, 2 removed lines. Original line number Diff line number Diff line Loading @@ -44,7 +44,7 @@ "device_operational_status": 1, "device_config": {"config_rules": [ {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2024"}}, {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": { "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false, "look_for_keys": false, "allow_agent": false, "commit_per_rule": false, Loading @@ -57,7 +57,7 @@ "device_operational_status": 1, "device_config": {"config_rules": [ {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2025"}}, {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": { "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false, "look_for_keys": false, "allow_agent": false, "commit_per_rule": false, Loading src/tests/ecoc26_pluggables/ecoc26_deploy.sh +27 −1 Changes for src/tests/ecoc26_pluggables/ecoc26_deploy.sh: 27 added lines, 1 removed line. Original line number Diff line number Diff line Loading @@ -99,7 +99,33 @@ export TFS_COMPONENTS="${TFS_COMPONENTS} kpi_manager telemetry analytics automat #export TFS_COMPONENTS="${TFS_COMPONENTS} load_generator" # Uncomment to activate Pluggables Component export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables" # export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables" #New functions to patch and restore the WebUI title, so that it can be used to indicate which test is running in the WebUI. # WEBUI_HOME_HTML must point at the real template file -- it was never set # before, so both sed calls below ran against an empty path. WEBUI_HOME_HTML="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)/src/webui/service/templates/main/home.html" patch_webui_title() { sed -i "s|\(<h2>ETSI TeraFlowSDN Controller\)</h2>|\1 ($1)</h2>|" "${WEBUI_HOME_HTML}" trap restore_webui_title EXIT } restore_webui_title() { trap - EXIT sed -i 's|\(<h2>ETSI TeraFlowSDN Controller\) ([^)]*)</h2>|\1</h2>|' "${WEBUI_HOME_HTML}" } # This file only exports env vars (deploy/all.sh is run afterward, as a # separate command, in the same sourced shell) -- calling restore_webui_title # here used to undo the patch within milliseconds, before that later deploy # step ever got a chance to build the webui image from the patched file. The # EXIT trap set inside patch_webui_title already restores the source file # once this shell session actually ends, so nothing else is needed here. patch_webui_title "Optical SDN Controller" # Set the tag you want to use for your images. Loading src/webui/service/templates/main/home.html +1 −1 Changes for src/webui/service/templates/main/home.html: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -17,7 +17,7 @@ {% extends 'base.html' %} {% block content %} <h2>ETSI TeraFlowSDN Controller</h2> <h2>ETSI TeraFlowSDN Controller (Optical SDN Controller)</h2> <style> #topology { border: 1px solid #b5b5b5; Loading Loading
src/opticalcontroller/Dockerfile +1 −1 Changes for src/opticalcontroller/Dockerfile: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -20,7 +20,7 @@ RUN apt-get --yes --quiet --quiet update && \ rm -rf /var/lib/apt/lists/* # Set Python to show logs as they occur ENV PYTHONUNBUFFERED=0 ENV PYTHONUNBUFFERED=1 # Download the gRPC health probe RUN GRPC_HEALTH_PROBE_VERSION=v0.2.0 && \ Loading
src/opticalcontroller/OpticalController.py +165 −9 Changes for src/opticalcontroller/OpticalController.py: 165 added lines, 9 removed lines. Original line number Diff line number Diff line Loading @@ -45,7 +45,6 @@ from google.protobuf.json_format import MessageToDict from common.tools.object_factory.OpticalLink import order_dict from opticalcontroller.RSA import RSA from opticalcontroller.RSA_P2MP import RSA_P2MP from opticalcontroller.dscm_monitoring import get_status_snapshot from common.tools.object_factory.Context import json_context_id from common.tools.object_factory.Topology import json_topology_id Loading Loading @@ -1222,8 +1221,21 @@ def process_topology(context_id: str, topology_id: str): OPTICAL_ROADM_TYPES = { DeviceTypeEnum.OPTICAL_ROADM.value, DeviceTypeEnum.EMULATED_OPTICAL_ROADM.value } # Includes PACKET_ROUTER alongside the real optical-transponder types: # ensure_topology.py deliberately registers the T1.1/T1.2/T2.1 # pluggables with device_type='packet-router' (not their descriptor's # real 'optical-transponder') so TelemetryBackendService's collector # filter -- which always matches on device_type=PACKET_ROUTER # regardless of the target device -- can find them. Without this, # those devices were silently excluded from node_dict below, so they # were never vertices in the RSA path-computation graph at all: every # link touching them got filtered out a few lines down, and # ComputeP2MP/PathRecomp always fell through to "No valid forward # path found" -- a hollow 201 with no subcarrier-attributes -- no # matter how the topology's links were configured. OPTICAL_TRANSPONDER_TYPES = { DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value, DeviceTypeEnum.PACKET_ROUTER.value, DeviceTypeEnum.EMULATED_PACKET_ROUTER.value, } added_device_uuids = set() Loading Loading @@ -1869,6 +1881,7 @@ class ComputeP2MP(Resource): _register_path_computation( sources, destinations, request_data, flow_id, via="ComputeP2MP" ) _maybe_revert_on_source_recompute(sources) LOGGER.info("******************************************") LOGGER.info("TAPI Path Computation successfully retrieved") LOGGER.info(f"TAPI Path Computation Response: {tapi_response}") Loading Loading @@ -2324,6 +2337,7 @@ class PathRecomp(ComputeP2MP): _register_path_computation( sources, destinations, request_data, flow_id, via="pathrecomp" ) _maybe_revert_on_source_recompute(sources) LOGGER.info("PathRecomp: new path registered for destinations=%s flow_id=%s", destinations, flow_id) Loading Loading @@ -2920,6 +2934,73 @@ def _start_leaf_monitoring(device_name: str) -> None: automation_client.close() # ── T1.2 mute/confirm/revert state machine ───────────────────────────────── # Hardcoded specifically for the T1.1<->T1.2 pair (not a generic per-device # mechanism -- the existing _confirmed_up_devices gate above already plays # that role for whichever device _LEAF_DEVICE_NAME currently is, and is left # untouched here). # # T1.2's own resting/quiet RECEIVED_POWER band -- distinct from T1.1's # _REARM_RECEIVED_POWER_RANGE_DBM above since it's a different physical # channel with its own baseline. Right after failover, T1.2 sitting in this # band with PRE_FEC_BER exactly 0.0 doesn't yet prove it's confirmed up -- # just that it hasn't started reporting real measurements yet -- so its # alarms are muted until a genuine (non-zero) in-range PRE_FEC_BER is seen # together with an in-range RECEIVED_POWER. Once confirmed, settling back to # PRE_FEC_BER=0.0 with power back in this same resting band is read as "the # T1.2 path is quiet again" and reverts monitoring back to T1.1. _T1_2_RESTING_POWER_RANGE_DBM = (-6.0, -3.0) # True once T1.2 has actually been confirmed up (see # _update_t1_2_confirmation_state) since the most recent failover to it -- # its alarms are muted until then. Reset to False every time a fresh # failover to T1.2 happens or monitoring reverts back to T1.1. _t1_2_confirmed_up = False def _update_t1_2_confirmation_state(device_name: str, kpi_name_upper: str, measured_value) -> None: """Run for every RECEIVED_POWER/PRE_FEC_BER sample PostAlarmNotification processes, but only actually does anything once T1.2 is the currently monitored device (i.e. a failover has already happened). Drives the mute -> confirm -> revert cycle described above; the caller in PostAlarmNotification is responsible for actually suppressing T1.2's alarms while not yet confirmed (see _t1_2_confirmed_up there).""" global _t1_2_confirmed_up if device_name != _LEAF_FAILOVER_DEVICE_NAME or _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME: return if "RECEIVED_POWER" not in kpi_name_upper and "PRE_FEC_BER" not in kpi_name_upper: return power = _last_received_power_by_device.get(device_name, (None, None))[0] ber = _last_pre_fec_ber_by_device.get(device_name) if power is None or ber is None: return # need both readings at least once before this can decide anything resting = _T1_2_RESTING_POWER_RANGE_DBM[0] <= power <= _T1_2_RESTING_POWER_RANGE_DBM[1] if not _t1_2_confirmed_up: if ber != 0.0 and PRE_FEC_BER_RANGE[0] <= ber <= PRE_FEC_BER_RANGE[1] \ and RECEIVED_POWER_RANGE_DBM[0] <= power <= RECEIVED_POWER_RANGE_DBM[1]: _t1_2_confirmed_up = True LOGGER.info( "T1.2 confirmed up: genuine in-range PRE_FEC_BER=%s with " "RECEIVED_POWER=%s -- alarm monitoring now trusted for T1.2.", ber, power, ) else: LOGGER.debug( "T1.2 not yet confirmed up (PRE_FEC_BER=%s, RECEIVED_POWER=%s) " "-- still muted.", ber, power, ) elif ber == 0.0 and resting: LOGGER.info( "T1.2 settled back to its resting baseline (PRE_FEC_BER=0.0, " "RECEIVED_POWER=%s in %s) -- reverting monitoring back to %s.", power, _T1_2_RESTING_POWER_RANGE_DBM, _ORIGINAL_LEAF_DEVICE_NAME, ) _revert_to_original_leaf_device() def _switch_monitored_device_and_recompute(notification: dict) -> None: """Called right after the first genuinely new alarm is actually delivered to a subscriber (see the two call sites below) -- replaces the old Loading @@ -2941,7 +3022,7 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None: Runs at most once: _leaf_failover_done latches after the switch, so the device's later alarms -- now resolved against the new monitored device -- don't re-trigger it, and the VOA is never pushed to again.""" global _LEAF_DEVICE_NAME, _leaf_failover_done global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up if _leaf_failover_done: return Loading Loading @@ -3013,6 +3094,44 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None: _LEAF_DEVICE_NAME = _LEAF_FAILOVER_DEVICE_NAME _leaf_failover_done = True _t1_2_confirmed_up = False def _revert_to_original_leaf_device() -> None: """Actually perform the revert from the failover device back to _ORIGINAL_LEAF_DEVICE_NAME: restart real monitoring on it and reset the failover latch/T1.2-confirmation state so a future alarm can trigger the same failover again from a clean baseline. Shared by _restore_monitored_device_after_delete (path-delete triggered), _update_t1_2_confirmation_state (T1.2 quiet-baseline triggered), and _maybe_revert_on_source_recompute (path-computation-sourced-from-T1.1 triggered).""" global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME) _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME _leaf_failover_done = False _t1_2_confirmed_up = False def _maybe_revert_on_source_recompute(sources) -> None: """If a path computation/recompute request names _ORIGINAL_LEAF_DEVICE_NAME (T1.1) as one of its sources while monitoring is currently on the failover device, that's a clear, explicit signal that T1.1 is back in service -- revert monitoring back to it immediately, the same as the path-delete and T1.2-quiet-baseline triggers do. Called from ComputeP2MP and PathRecomp (which also covers /pathrecomp and /recompute-p2mp) right after either one successfully registers a new path.""" if _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME: return # not currently failed over -- nothing to revert if _ORIGINAL_LEAF_DEVICE_NAME not in (sources or []): return LOGGER.info( "Path computation sourced from %s while monitoring %s -- " "reverting monitoring back to %s.", _ORIGINAL_LEAF_DEVICE_NAME, _LEAF_FAILOVER_DEVICE_NAME, _ORIGINAL_LEAF_DEVICE_NAME, ) _revert_to_original_leaf_device() def _restore_monitored_device_after_delete(deleted: dict) -> None: Loading Loading @@ -3043,9 +3162,7 @@ def _restore_monitored_device_after_delete(deleted: dict) -> None: "restoring monitoring to %s.", failed_over_device, _ORIGINAL_LEAF_DEVICE_NAME, ) _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME) _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME _leaf_failover_done = False _revert_to_original_leaf_device() def _push_notification_to_webhook(notification: dict) -> None: Loading @@ -3061,7 +3178,7 @@ def _push_notification_to_webhook(notification: dict) -> None: try: # The receiver validates against the full TAPI notification-service # schema, which wraps each notification in a "notification-context" # array (see tests/ecoc26_pluggables/tests/test_e2e.py's # array (see tests/ecoc26_pluggables/tests/.py's # DEFAULT_PAYLOAD) -- posting the bare "...notification" object by # itself is rejected with a 400 "required property" error. resp = requests.post( Loading Loading @@ -3279,13 +3396,30 @@ class PostAlarmNotification(Resource): # Baseline-health gate (see _record_breach_state_and_check_confirmed_up # above): suppress every alarm for this device until PRE_FEC_BER and # RECEIVED_POWER have both been observed in range at once. # # The gate is fed gate_low_breached/gate_high_breached rather than # the raw low_breached/high_breached: a raw PRE_FEC_BER==0.0 fails # its own range check (0.0 is below PRE_FEC_BER_RANGE's floor), but # 0.0 is the best possible BER reading on its own (see # is_pre_fec_ber_zero above) -- treating it as a breach here meant a # device whose healthy baseline is a clean 0.0 could NEVER satisfy # the gate, leaving it (and every real fault on it afterward) muted # forever. This correction only affects what the gate records -- # the actual alarm decision below (including the RECEIVED_POWER # correlation that can still escalate a 0.0 reading to a real # failure) keeps using the original low_breached/high_breached, # unaffected by this. gate_low_breached, gate_high_breached = low_breached, high_breached if is_pre_fec_ber_zero: gate_low_breached, gate_high_breached = False, False metric = ("PRE_FEC_BER" if "PRE_FEC_BER" in kpi_name_upper else "RECEIVED_POWER" if "RECEIVED_POWER" in kpi_name_upper else None) confirmed_up = True if metric is not None: confirmed_up = _record_breach_state_and_check_confirmed_up( device_name, metric, low_breached, high_breached) device_name, metric, gate_low_breached, gate_high_breached) if not confirmed_up: # Silent while unconfirmed -- this fires on every single # sample until the device is confirmed up, so it stays at Loading Loading @@ -3335,6 +3469,28 @@ class PostAlarmNotification(Resource): ) low_breached = False # T1.2-specific mute/confirm/revert state machine (see # _update_t1_2_confirmation_state above) -- runs independently of, # and after, the generic confirmed-up gate above. Only actually # takes effect once T1.2 is the currently monitored device (i.e. # after a failover); a no-op otherwise. Has the final say for T1.2's # own alarms: even if the checks above decided this is a genuine # breach, T1.2 stays muted until it's been confirmed up in its own # right. if metric is not None and device_name: _update_t1_2_confirmation_state(device_name, kpi_name_upper, measured_value) if (device_name == _LEAF_FAILOVER_DEVICE_NAME and _LEAF_DEVICE_NAME == _LEAF_FAILOVER_DEVICE_NAME and not _t1_2_confirmed_up): LOGGER.debug( "T1.2 not yet confirmed up -- suppressing alarm for service %s / " "KPI %s (value=%s).", data.get("service_id", ""), kpi_context["kpi-name"], measured_value, ) g.suppress_latency_log = True low_breached = False high_breached = False if not low_breached and not high_breached: # DEBUG, not INFO: this fires on every single non-breaching # sample (i.e. most of them, most of the time) -- only a Loading Loading @@ -3380,7 +3536,7 @@ class PostAlarmNotification(Resource): "probable-cause": "THRESHOLD_CROSSED", # The receiver's TAPI schema requires these as numbers, not # booleans (validated against tests/ecoc26_pluggables/tests/ # test_e2e.py's DEFAULT_PAYLOAD) -- policyservice only tells # .py's DEFAULT_PAYLOAD) -- policyservice only tells # us whether each threshold was breached, not the KPI's # actual configured threshold value, so 1.0/0.0 preserves # that breach flag as a number instead of fabricating a Loading
src/tests/PMPv1/descriptors/singleEndpoint.json +2 −2 Changes for src/tests/PMPv1/descriptors/singleEndpoint.json: 2 added lines, 2 removed lines. Original line number Diff line number Diff line Loading @@ -44,7 +44,7 @@ "device_operational_status": 1, "device_config": {"config_rules": [ {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2024"}}, {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": { "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false, "look_for_keys": false, "allow_agent": false, "commit_per_rule": false, Loading @@ -57,7 +57,7 @@ "device_operational_status": 1, "device_config": {"config_rules": [ {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}}, {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2025"}}, {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": { "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false, "look_for_keys": false, "allow_agent": false, "commit_per_rule": false, Loading
src/tests/ecoc26_pluggables/ecoc26_deploy.sh +27 −1 Changes for src/tests/ecoc26_pluggables/ecoc26_deploy.sh: 27 added lines, 1 removed line. Original line number Diff line number Diff line Loading @@ -99,7 +99,33 @@ export TFS_COMPONENTS="${TFS_COMPONENTS} kpi_manager telemetry analytics automat #export TFS_COMPONENTS="${TFS_COMPONENTS} load_generator" # Uncomment to activate Pluggables Component export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables" # export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables" #New functions to patch and restore the WebUI title, so that it can be used to indicate which test is running in the WebUI. # WEBUI_HOME_HTML must point at the real template file -- it was never set # before, so both sed calls below ran against an empty path. WEBUI_HOME_HTML="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)/src/webui/service/templates/main/home.html" patch_webui_title() { sed -i "s|\(<h2>ETSI TeraFlowSDN Controller\)</h2>|\1 ($1)</h2>|" "${WEBUI_HOME_HTML}" trap restore_webui_title EXIT } restore_webui_title() { trap - EXIT sed -i 's|\(<h2>ETSI TeraFlowSDN Controller\) ([^)]*)</h2>|\1</h2>|' "${WEBUI_HOME_HTML}" } # This file only exports env vars (deploy/all.sh is run afterward, as a # separate command, in the same sourced shell) -- calling restore_webui_title # here used to undo the patch within milliseconds, before that later deploy # step ever got a chance to build the webui image from the patched file. The # EXIT trap set inside patch_webui_title already restores the source file # once this shell session actually ends, so nothing else is needed here. patch_webui_title "Optical SDN Controller" # Set the tag you want to use for your images. Loading
src/webui/service/templates/main/home.html +1 −1 Changes for src/webui/service/templates/main/home.html: 1 added line, 1 removed line. Original line number Diff line number Diff line Loading @@ -17,7 +17,7 @@ {% extends 'base.html' %} {% block content %} <h2>ETSI TeraFlowSDN Controller</h2> <h2>ETSI TeraFlowSDN Controller (Optical SDN Controller)</h2> <style> #topology { border: 1px solid #b5b5b5; Loading