Commit 18e17669 authored by Andrea Sgambelluri's avatar Andrea Sgambelluri
Browse files

Correction to reduce latency in alarm

parent a4e1490d
Loading
Loading
Loading
Loading
+1 −1
Changes for src/opticalcontroller/Dockerfile: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -20,7 +20,7 @@ RUN apt-get --yes --quiet --quiet update && \
    rm -rf /var/lib/apt/lists/*

# Set Python to show logs as they occur
ENV PYTHONUNBUFFERED=0
ENV PYTHONUNBUFFERED=1

# Download the gRPC health probe
RUN GRPC_HEALTH_PROBE_VERSION=v0.2.0 && \
+165 −9
Changes for src/opticalcontroller/OpticalController.py: 165 added lines, 9 removed lines.
Original line number Diff line number Diff line
@@ -45,7 +45,6 @@ from google.protobuf.json_format import MessageToDict
from common.tools.object_factory.OpticalLink import order_dict
from opticalcontroller.RSA import RSA
from opticalcontroller.RSA_P2MP import RSA_P2MP
from opticalcontroller.dscm_monitoring import get_status_snapshot

from common.tools.object_factory.Context import json_context_id
from common.tools.object_factory.Topology import json_topology_id
@@ -1222,8 +1221,21 @@ def process_topology(context_id: str, topology_id: str):
        OPTICAL_ROADM_TYPES = {
            DeviceTypeEnum.OPTICAL_ROADM.value, DeviceTypeEnum.EMULATED_OPTICAL_ROADM.value
        }
        # Includes PACKET_ROUTER alongside the real optical-transponder types:
        # ensure_topology.py deliberately registers the T1.1/T1.2/T2.1
        # pluggables with device_type='packet-router' (not their descriptor's
        # real 'optical-transponder') so TelemetryBackendService's collector
        # filter -- which always matches on device_type=PACKET_ROUTER
        # regardless of the target device -- can find them. Without this,
        # those devices were silently excluded from node_dict below, so they
        # were never vertices in the RSA path-computation graph at all: every
        # link touching them got filtered out a few lines down, and
        # ComputeP2MP/PathRecomp always fell through to "No valid forward
        # path found" -- a hollow 201 with no subcarrier-attributes -- no
        # matter how the topology's links were configured.
        OPTICAL_TRANSPONDER_TYPES = {
            DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value
            DeviceTypeEnum.OPTICAL_TRANSPONDER.value, DeviceTypeEnum.EMULATED_OPTICAL_TRANSPONDER.value,
            DeviceTypeEnum.PACKET_ROUTER.value, DeviceTypeEnum.EMULATED_PACKET_ROUTER.value,
        }

        added_device_uuids = set()
@@ -1869,6 +1881,7 @@ class ComputeP2MP(Resource):
                _register_path_computation(
                    sources, destinations, request_data, flow_id, via="ComputeP2MP"
                )
                _maybe_revert_on_source_recompute(sources)
                LOGGER.info("******************************************")
                LOGGER.info("TAPI Path Computation successfully retrieved")
                LOGGER.info(f"TAPI Path Computation Response: {tapi_response}")
@@ -2324,6 +2337,7 @@ class PathRecomp(ComputeP2MP):
                _register_path_computation(
                    sources, destinations, request_data, flow_id, via="pathrecomp"
                )
                _maybe_revert_on_source_recompute(sources)
                LOGGER.info("PathRecomp: new path registered for destinations=%s flow_id=%s",
                            destinations, flow_id)

@@ -2920,6 +2934,73 @@ def _start_leaf_monitoring(device_name: str) -> None:
            automation_client.close()


# ── T1.2 mute/confirm/revert state machine ─────────────────────────────────
# Hardcoded specifically for the T1.1<->T1.2 pair (not a generic per-device
# mechanism -- the existing _confirmed_up_devices gate above already plays
# that role for whichever device _LEAF_DEVICE_NAME currently is, and is left
# untouched here).
#
# T1.2's own resting/quiet RECEIVED_POWER band -- distinct from T1.1's
# _REARM_RECEIVED_POWER_RANGE_DBM above since it's a different physical
# channel with its own baseline. Right after failover, T1.2 sitting in this
# band with PRE_FEC_BER exactly 0.0 doesn't yet prove it's confirmed up --
# just that it hasn't started reporting real measurements yet -- so its
# alarms are muted until a genuine (non-zero) in-range PRE_FEC_BER is seen
# together with an in-range RECEIVED_POWER. Once confirmed, settling back to
# PRE_FEC_BER=0.0 with power back in this same resting band is read as "the
# T1.2 path is quiet again" and reverts monitoring back to T1.1.
_T1_2_RESTING_POWER_RANGE_DBM = (-6.0, -3.0)

# True once T1.2 has actually been confirmed up (see
# _update_t1_2_confirmation_state) since the most recent failover to it --
# its alarms are muted until then. Reset to False every time a fresh
# failover to T1.2 happens or monitoring reverts back to T1.1.
_t1_2_confirmed_up = False


def _update_t1_2_confirmation_state(device_name: str, kpi_name_upper: str, measured_value) -> None:
    """Run for every RECEIVED_POWER/PRE_FEC_BER sample PostAlarmNotification
    processes, but only actually does anything once T1.2 is the currently
    monitored device (i.e. a failover has already happened). Drives the
    mute -> confirm -> revert cycle described above; the caller in
    PostAlarmNotification is responsible for actually suppressing T1.2's
    alarms while not yet confirmed (see _t1_2_confirmed_up there)."""
    global _t1_2_confirmed_up
    if device_name != _LEAF_FAILOVER_DEVICE_NAME or _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME:
        return
    if "RECEIVED_POWER" not in kpi_name_upper and "PRE_FEC_BER" not in kpi_name_upper:
        return

    power = _last_received_power_by_device.get(device_name, (None, None))[0]
    ber   = _last_pre_fec_ber_by_device.get(device_name)
    if power is None or ber is None:
        return  # need both readings at least once before this can decide anything

    resting = _T1_2_RESTING_POWER_RANGE_DBM[0] <= power <= _T1_2_RESTING_POWER_RANGE_DBM[1]

    if not _t1_2_confirmed_up:
        if ber != 0.0 and PRE_FEC_BER_RANGE[0] <= ber <= PRE_FEC_BER_RANGE[1] \
                and RECEIVED_POWER_RANGE_DBM[0] <= power <= RECEIVED_POWER_RANGE_DBM[1]:
            _t1_2_confirmed_up = True
            LOGGER.info(
                "T1.2 confirmed up: genuine in-range PRE_FEC_BER=%s with "
                "RECEIVED_POWER=%s -- alarm monitoring now trusted for T1.2.",
                ber, power,
            )
        else:
            LOGGER.debug(
                "T1.2 not yet confirmed up (PRE_FEC_BER=%s, RECEIVED_POWER=%s) "
                "-- still muted.", ber, power,
            )
    elif ber == 0.0 and resting:
        LOGGER.info(
            "T1.2 settled back to its resting baseline (PRE_FEC_BER=0.0, "
            "RECEIVED_POWER=%s in %s) -- reverting monitoring back to %s.",
            power, _T1_2_RESTING_POWER_RANGE_DBM, _ORIGINAL_LEAF_DEVICE_NAME,
        )
        _revert_to_original_leaf_device()


def _switch_monitored_device_and_recompute(notification: dict) -> None:
    """Called right after the first genuinely new alarm is actually delivered
    to a subscriber (see the two call sites below) -- replaces the old
@@ -2941,7 +3022,7 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None:
    Runs at most once: _leaf_failover_done latches after the switch, so the
    device's later alarms -- now resolved against the new monitored device
    -- don't re-trigger it, and the VOA is never pushed to again."""
    global _LEAF_DEVICE_NAME, _leaf_failover_done
    global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up
    if _leaf_failover_done:
        return

@@ -3013,6 +3094,44 @@ def _switch_monitored_device_and_recompute(notification: dict) -> None:

    _LEAF_DEVICE_NAME = _LEAF_FAILOVER_DEVICE_NAME
    _leaf_failover_done = True
    _t1_2_confirmed_up = False


def _revert_to_original_leaf_device() -> None:
    """Actually perform the revert from the failover device back to
    _ORIGINAL_LEAF_DEVICE_NAME: restart real monitoring on it and reset the
    failover latch/T1.2-confirmation state so a future alarm can trigger the
    same failover again from a clean baseline. Shared by
    _restore_monitored_device_after_delete (path-delete triggered),
    _update_t1_2_confirmation_state (T1.2 quiet-baseline triggered), and
    _maybe_revert_on_source_recompute (path-computation-sourced-from-T1.1
    triggered)."""
    global _LEAF_DEVICE_NAME, _leaf_failover_done, _t1_2_confirmed_up
    _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME)
    _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME
    _leaf_failover_done = False
    _t1_2_confirmed_up = False


def _maybe_revert_on_source_recompute(sources) -> None:
    """If a path computation/recompute request names _ORIGINAL_LEAF_DEVICE_NAME
    (T1.1) as one of its sources while monitoring is currently on the
    failover device, that's a clear, explicit signal that T1.1 is back in
    service -- revert monitoring back to it immediately, the same as the
    path-delete and T1.2-quiet-baseline triggers do. Called from ComputeP2MP
    and PathRecomp (which also covers /pathrecomp and /recompute-p2mp) right
    after either one successfully registers a new path."""
    if _LEAF_DEVICE_NAME != _LEAF_FAILOVER_DEVICE_NAME:
        return  # not currently failed over -- nothing to revert
    if _ORIGINAL_LEAF_DEVICE_NAME not in (sources or []):
        return

    LOGGER.info(
        "Path computation sourced from %s while monitoring %s -- "
        "reverting monitoring back to %s.",
        _ORIGINAL_LEAF_DEVICE_NAME, _LEAF_FAILOVER_DEVICE_NAME, _ORIGINAL_LEAF_DEVICE_NAME,
    )
    _revert_to_original_leaf_device()


def _restore_monitored_device_after_delete(deleted: dict) -> None:
@@ -3043,9 +3162,7 @@ def _restore_monitored_device_after_delete(deleted: dict) -> None:
        "restoring monitoring to %s.",
        failed_over_device, _ORIGINAL_LEAF_DEVICE_NAME,
    )
    _start_leaf_monitoring(_ORIGINAL_LEAF_DEVICE_NAME)
    _LEAF_DEVICE_NAME = _ORIGINAL_LEAF_DEVICE_NAME
    _leaf_failover_done = False
    _revert_to_original_leaf_device()


def _push_notification_to_webhook(notification: dict) -> None:
@@ -3061,7 +3178,7 @@ def _push_notification_to_webhook(notification: dict) -> None:
    try:
        # The receiver validates against the full TAPI notification-service
        # schema, which wraps each notification in a "notification-context"
        # array (see tests/ecoc26_pluggables/tests/test_e2e.py's
        # array (see tests/ecoc26_pluggables/tests/.py's
        # DEFAULT_PAYLOAD) -- posting the bare "...notification" object by
        # itself is rejected with a 400 "required property" error.
        resp = requests.post(
@@ -3279,13 +3396,30 @@ class PostAlarmNotification(Resource):
        # Baseline-health gate (see _record_breach_state_and_check_confirmed_up
        # above): suppress every alarm for this device until PRE_FEC_BER and
        # RECEIVED_POWER have both been observed in range at once.
        #
        # The gate is fed gate_low_breached/gate_high_breached rather than
        # the raw low_breached/high_breached: a raw PRE_FEC_BER==0.0 fails
        # its own range check (0.0 is below PRE_FEC_BER_RANGE's floor), but
        # 0.0 is the best possible BER reading on its own (see
        # is_pre_fec_ber_zero above) -- treating it as a breach here meant a
        # device whose healthy baseline is a clean 0.0 could NEVER satisfy
        # the gate, leaving it (and every real fault on it afterward) muted
        # forever. This correction only affects what the gate records --
        # the actual alarm decision below (including the RECEIVED_POWER
        # correlation that can still escalate a 0.0 reading to a real
        # failure) keeps using the original low_breached/high_breached,
        # unaffected by this.
        gate_low_breached, gate_high_breached = low_breached, high_breached
        if is_pre_fec_ber_zero:
            gate_low_breached, gate_high_breached = False, False

        metric = ("PRE_FEC_BER" if "PRE_FEC_BER" in kpi_name_upper
                  else "RECEIVED_POWER" if "RECEIVED_POWER" in kpi_name_upper
                  else None)
        confirmed_up = True
        if metric is not None:
            confirmed_up = _record_breach_state_and_check_confirmed_up(
                device_name, metric, low_breached, high_breached)
                device_name, metric, gate_low_breached, gate_high_breached)
            if not confirmed_up:
                # Silent while unconfirmed -- this fires on every single
                # sample until the device is confirmed up, so it stays at
@@ -3335,6 +3469,28 @@ class PostAlarmNotification(Resource):
                )
                low_breached = False

        # T1.2-specific mute/confirm/revert state machine (see
        # _update_t1_2_confirmation_state above) -- runs independently of,
        # and after, the generic confirmed-up gate above. Only actually
        # takes effect once T1.2 is the currently monitored device (i.e.
        # after a failover); a no-op otherwise. Has the final say for T1.2's
        # own alarms: even if the checks above decided this is a genuine
        # breach, T1.2 stays muted until it's been confirmed up in its own
        # right.
        if metric is not None and device_name:
            _update_t1_2_confirmation_state(device_name, kpi_name_upper, measured_value)
            if (device_name == _LEAF_FAILOVER_DEVICE_NAME
                    and _LEAF_DEVICE_NAME == _LEAF_FAILOVER_DEVICE_NAME
                    and not _t1_2_confirmed_up):
                LOGGER.debug(
                    "T1.2 not yet confirmed up -- suppressing alarm for service %s / "
                    "KPI %s (value=%s).",
                    data.get("service_id", ""), kpi_context["kpi-name"], measured_value,
                )
                g.suppress_latency_log = True
                low_breached = False
                high_breached = False

        if not low_breached and not high_breached:
            # DEBUG, not INFO: this fires on every single non-breaching
            # sample (i.e. most of them, most of the time) -- only a
@@ -3380,7 +3536,7 @@ class PostAlarmNotification(Resource):
                    "probable-cause":         "THRESHOLD_CROSSED",
                    # The receiver's TAPI schema requires these as numbers, not
                    # booleans (validated against tests/ecoc26_pluggables/tests/
                    # test_e2e.py's DEFAULT_PAYLOAD) -- policyservice only tells
                    # .py's DEFAULT_PAYLOAD) -- policyservice only tells
                    # us whether each threshold was breached, not the KPI's
                    # actual configured threshold value, so 1.0/0.0 preserves
                    # that breach flag as a number instead of fabricating a
+2 −2
Changes for src/tests/PMPv1/descriptors/singleEndpoint.json: 2 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -44,7 +44,7 @@
            "device_operational_status": 1,
            "device_config": {"config_rules": [
                {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}},
                {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}},
                {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2024"}},
                {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": {
                    "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false,
                    "look_for_keys": false, "allow_agent": false, "commit_per_rule": false,
@@ -57,7 +57,7 @@
            "device_operational_status": 1,
            "device_config": {"config_rules": [
                {"action": 1, "custom": {"resource_key": "_connect/address", "resource_value": "10.30.7.65"}},
                {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2023"}},
                {"action": 1, "custom": {"resource_key": "_connect/port", "resource_value": "2025"}},
                {"action": 1, "custom": {"resource_key": "_connect/settings", "resource_value": {
                    "username": "admin", "password": "admin", "force_running": false, "hostkey_verify": false,
                    "look_for_keys": false, "allow_agent": false, "commit_per_rule": false,
+27 −1
Changes for src/tests/ecoc26_pluggables/ecoc26_deploy.sh: 27 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -99,7 +99,33 @@ export TFS_COMPONENTS="${TFS_COMPONENTS} kpi_manager telemetry analytics automat
#export TFS_COMPONENTS="${TFS_COMPONENTS} load_generator"

# Uncomment to activate Pluggables Component
export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables"
# export TFS_COMPONENTS="${TFS_COMPONENTS} pluggables"



#New functions to patch and restore the WebUI title, so that it can be used to indicate which test is running in the WebUI.
# WEBUI_HOME_HTML must point at the real template file -- it was never set
# before, so both sed calls below ran against an empty path.
WEBUI_HOME_HTML="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)/src/webui/service/templates/main/home.html"

patch_webui_title() {
    sed -i "s|\(<h2>ETSI TeraFlowSDN Controller\)</h2>|\1 ($1)</h2>|" "${WEBUI_HOME_HTML}"
    trap restore_webui_title EXIT
}

restore_webui_title() {
    trap - EXIT
    sed -i 's|\(<h2>ETSI TeraFlowSDN Controller\) ([^)]*)</h2>|\1</h2>|' "${WEBUI_HOME_HTML}"
}


# This file only exports env vars (deploy/all.sh is run afterward, as a
# separate command, in the same sourced shell) -- calling restore_webui_title
# here used to undo the patch within milliseconds, before that later deploy
# step ever got a chance to build the webui image from the patched file. The
# EXIT trap set inside patch_webui_title already restores the source file
# once this shell session actually ends, so nothing else is needed here.
patch_webui_title "Optical SDN Controller"


# Set the tag you want to use for your images.
+1 −1
Changes for src/webui/service/templates/main/home.html: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -17,7 +17,7 @@
{% extends 'base.html' %}

{% block content %}
    <h2>ETSI TeraFlowSDN Controller</h2>
    <h2>ETSI TeraFlowSDN Controller (Optical SDN Controller)</h2>
    <style>
        #topology {
            border: 1px solid #b5b5b5;
Loading