From 798438d7a64826976e8d543f5a011e780a71f631 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Fri, 31 Jul 2026 20:41:31 +0530 Subject: [PATCH 01/28] UN-3494 [FEAT] Email group members on resource share and group membership changes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sharing a resource with a group gave its members access silently, and adding or removing someone from a group told nobody. Both now send email. - share_notifications.py holds the feature flag, the two task names and the two enqueue hooks. Dispatch uses the same resolve_transport branch the execution path uses: the PG queue where pg_queue_enabled is on for the org, Celery otherwise. - One hook in ResourceShareManagementMixin.share covers all 7 resource types plus cloud agentic, including service-account shares — every group share funnels through it and shared_groups has no PATCH path. No on_commit needed: _commit's transaction has closed by the time the view resumes, so the diff reads committed state. - Group membership hooks on the add and remove actions. The add serializer already subtracts existing members, so nobody is mailed twice. - Internal endpoints under /internal/v1/group-notification/ do the work the worker cannot: group expansion, OrganizationMember re-validation (this is where the offboarding race closes), resource lookup via ShareableResource, and the kind -> ResourceType mapping, which is not 1:1 — pipelines split on pipeline_type and adapters four ways on adapter_type. - Two worker tasks that only POST to that endpoint, since workers/ has no Django. They raise on failure, unlike _mark_buffer_outcome which has a reaper behind it, and retry transient 5xx in-task because a raise is terminal on the Celery transport. - The whole feature is gated on Flipt group_sharing_notifications_enabled and fails closed: a blind Flipt, a missing org, or any dispatch error means no notification, never a broken share. - worker-pg-notification compose service so the PG arm is not a black hole. Membership changes with no actor (the org-removal cascade, Django admin, group deletion) do not notify — SharingNotificationService requires an actor. Co-Authored-By: Claude Opus 5 --- backend/backend/internal_base_urls.py | 6 + backend/permissions/resource_share_views.py | 16 ++ .../group_notification_service.py | 237 ++++++++++++++++++ backend/tenant_account_v2/group_views.py | 18 ++ backend/tenant_account_v2/internal_urls.py | 20 ++ backend/tenant_account_v2/internal_views.py | 91 +++++++ .../tenant_account_v2/share_notifications.py | 210 ++++++++++++++++ .../tenant_account_v2/shareable_resources.py | 24 ++ docker/docker-compose.yaml | 36 +++ workers/notification/tasks.py | 99 ++++++++ 10 files changed, 757 insertions(+) create mode 100644 backend/tenant_account_v2/group_notification_service.py create mode 100644 backend/tenant_account_v2/internal_urls.py create mode 100644 backend/tenant_account_v2/internal_views.py create mode 100644 backend/tenant_account_v2/share_notifications.py diff --git a/backend/backend/internal_base_urls.py b/backend/backend/internal_base_urls.py index 0354a691ae..e89c5519f1 100644 --- a/backend/backend/internal_base_urls.py +++ b/backend/backend/internal_base_urls.py @@ -269,4 +269,10 @@ def test_middleware_debug(request): include("prompt_studio.prompt_studio_core_v2.internal_urls"), name="prompt_studio_internal", ), + # Group-sharing email notification APIs + path( + "v1/group-notification/", + include("tenant_account_v2.internal_urls"), + name="group_notification_internal", + ), ] diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 1227562531..f688569115 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -91,13 +91,29 @@ def share(self, request: Request, pk: str | None = None) -> Response: users, group-membership for groups) live in ``ShareAuthorizationService``. """ + from tenant_account_v2.share_notifications import ( + notify_resource_shared_with_group, + ) from tenant_account_v2.sharing_helpers import ShareAuthorizationService resource = self.get_object() # type: ignore[attr-defined] desired = _extract_desired_share_state(request.data) + # Only the groups axis is diffed: it is the one that notifies, and + # snapshotting ``shared_users`` too would fetch every viewer twice for + # nothing. Reads go through ``ResourceGroupShare``, so no refresh is + # needed between the two. + groups_before = self._read_axis(resource, "shared_groups") ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) + # ``_commit`` is the only atomic block on this path, so it has already + # committed — the diff reads persisted state and can never announce a + # share that rolled back. + notify_resource_shared_with_group( + resource=resource, + groups=self._read_axis(resource, "shared_groups") - groups_before, + actor=request.user, + ) return Response(status=status.HTTP_200_OK) @action(detail=True, methods=["get"], url_path="effective-members") diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py new file mode 100644 index 0000000000..53f2bc7f82 --- /dev/null +++ b/backend/tenant_account_v2/group_notification_service.py @@ -0,0 +1,237 @@ +"""Send-side logic for group-sharing email notifications (UN-3494 / mfbt UNS-848). + +Reached over the internal API by the notification worker. The enqueue side +(:mod:`tenant_account_v2.share_notifications`) only records *what happened*; +everything that needs Django — group expansion, org re-validation, resource +lookup, the email plugin — happens here, because ``workers/`` has no Django. + +Sending is a cloud plugin. In OSS ``notification_plugin`` is empty and every +entry point below no-ops cleanly. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING, Any + +from account_v2.models import Organization, User +from django.apps import apps +from django.db.models import QuerySet +from plugins import get_plugin + +from tenant_account_v2.models import OrganizationGroup, OrganizationMember +from tenant_account_v2.share_notifications import MembershipAction +from tenant_account_v2.shareable_resources import ShareableResource, descriptor_for_kind + +if TYPE_CHECKING: + from collections.abc import Iterable + +logger = logging.getLogger(__name__) + +notification_plugin = get_plugin("notification") + +# OSS ``ShareableResource.kind`` → the email plugin's ``ResourceType`` value. +# Deliberately plain strings: OSS must not import a cloud-only enum. Not a 1:1 +# rename — pipelines and adapters resolve from the instance below. +_STATIC_RESOURCE_TYPES = { + "workflow": "workflow", + "api_deployment": "api", + "connector_instance": "connector", + "custom_tool": "text_extractor", + "agentic_project": "agentic_project", +} +_ADAPTER_RESOURCE_TYPES = { + "LLM": "llm", + "EMBEDDING": "embedding", + "VECTOR_DB": "vector_db", + "X2TEXT": "x2text", +} +# Only ETL/TASK pipelines map to a notification resource type; the plugin +# compares against these exact (uppercase) values. +_PIPELINE_RESOURCE_TYPES = frozenset({"ETL", "TASK"}) + + +class ResourceNotFoundError(Exception): + """The shared resource no longer exists, or is not in the given org.""" + + +def send_resource_shared( + *, + organization: Organization, + group_ids: Iterable[int], + actor_id: int, + resource_kind: str, + resource_id: str, +) -> None: + """Mail every current member of each group that a resource was shared. + + One email per group, so ``group_name`` in the template is always the group + the recipient actually belongs to. + """ + service = _service() + if service is None: + return + actor = _get_user(actor_id) + resource, resource_name, resource_type = _load_resource( + organization, resource_kind, resource_id + ) + if actor is None or resource_type is None: + logger.info( + "group-notification: skipping resource share for %s/%s " + "(actor_found=%s resource_type=%s)", + resource_kind, + resource_id, + actor is not None, + resource_type, + ) + return + for group in _groups_in_org(organization, group_ids): + recipients = _live_member_users( + organization, group.memberships.values_list("user_id", flat=True) + ) + logger.info( + "group-notification: task=%s group_id=%s recipient_count=%d", + "notify_resource_shared_with_group", + group.pk, + len(recipients), + ) + if not recipients: + continue + service.send_group_resource_shared_notification( + resource_type=resource_type, + resource_name=resource_name, + resource_id=str(resource.pk), + group_name=group.name, + shared_by=actor, + shared_to=recipients, + resource_instance=resource, + ) + + +def send_membership_changed( + *, + organization: Organization, + group_id: int, + actor_id: int, + membership_action: str, + user_ids: Iterable[int], +) -> None: + """Mail the users whose membership of ``group_id`` just changed. + + Recipients are re-validated against ``OrganizationMember`` — this is where + the offboarding race closes, for removals as well as additions: leaving a + group does not remove someone from the org, so both directions validate the + same way. + """ + service = _service() + if service is None: + return + actor = _get_user(actor_id) + group = _groups_in_org(organization, [group_id]).first() + if actor is None or group is None: + logger.info( + "group-notification: skipping membership change for group %s " + "(actor_found=%s group_found=%s)", + group_id, + actor is not None, + group is not None, + ) + return + recipients = _live_member_users(organization, user_ids) + logger.info( + "group-notification: task=%s group_id=%s action=%s recipient_count=%d", + "notify_group_membership_changed", + group.pk, + membership_action, + len(recipients), + ) + if not recipients: + return + service.send_group_membership_notification( + group_name=group.name, + membership_action=MembershipAction(membership_action).value, + recipients=recipients, + actor=actor, + organization=organization, + ) + + +def _service() -> Any | None: + """The cloud email service, or ``None`` when the plugin is absent (OSS).""" + if not notification_plugin: + logger.debug("group-notification: notification plugin unavailable, skipping") + return None + return notification_plugin["service_class"]() + + +def _get_user(user_id: int) -> User | None: + return User.objects.filter(pk=user_id).first() + + +def _groups_in_org( + organization: Organization, group_ids: Iterable[int] +) -> QuerySet[OrganizationGroup]: + """Groups from ``group_ids`` that belong to ``organization``.""" + return OrganizationGroup.objects.filter( + organization=organization, pk__in=list(group_ids) + ) + + +def _live_member_users(organization: Organization, user_ids: Iterable[int]) -> list[User]: + """Users from ``user_ids`` who are still live members of ``organization``. + + Service accounts are excluded, matching ``compute_effective_members``. + """ + memberships = OrganizationMember.objects.filter( + organization=organization, user_id__in=list(user_ids) + ).select_related("user") + return [ + m.user + for m in memberships + if not getattr(m.user, "is_service_account", False) and m.user.email + ] + + +def _load_resource( + organization: Organization, kind: str, resource_id: str +) -> tuple[Any, str, str | None]: + """Resolve the shared resource to ``(instance, display name, plugin type)``. + + Raises: + ResourceNotFoundError: the descriptor, model, or row is missing — the + resource was deleted or belongs to another org. Callers turn this + into a success so the queue stops retrying. + """ + descriptor = descriptor_for_kind(kind) + if descriptor is None: + raise ResourceNotFoundError(f"Unknown resource kind: {kind}") + try: + model = apps.get_model(descriptor.app_label, descriptor.model_name) + except LookupError as exc: # cloud-only app not installed here + raise ResourceNotFoundError(f"Model unavailable for kind: {kind}") from exc + # Filter on the organization explicitly rather than trusting the default + # manager: ``AgenticProject``'s manager deliberately spans organizations. + resource = model.objects.filter( + organization=organization, **{descriptor.id_field: resource_id} + ).first() + if resource is None: + raise ResourceNotFoundError(f"{kind} {resource_id} not found in organization") + name = getattr(resource, descriptor.name_field, "") or "" + return resource, name, _resource_type_for(descriptor, resource) + + +def _resource_type_for(descriptor: ShareableResource, resource: Any) -> str | None: + """Map a resource to the email plugin's ``ResourceType`` value. + + Returns ``None`` for resources the plugin has no type for (e.g. a pipeline + that is neither ETL nor TASK) — the caller skips rather than guessing. + """ + if descriptor.kind == "pipeline": + pipeline_type = getattr(resource, "pipeline_type", None) + return pipeline_type if pipeline_type in _PIPELINE_RESOURCE_TYPES else None + if descriptor.kind == "adapter_instance": + # Unknown adapter types fall back to ``llm``, matching the co-owner + # path's override — an OCR adapter shared with a group should not + # silently send nothing when sharing it with a co-owner mails fine. + return _ADAPTER_RESOURCE_TYPES.get(str(resource.adapter_type or ""), "llm") + return _STATIC_RESOURCE_TYPES.get(descriptor.kind) diff --git a/backend/tenant_account_v2/group_views.py b/backend/tenant_account_v2/group_views.py index 76461951fa..d5013b81b8 100644 --- a/backend/tenant_account_v2/group_views.py +++ b/backend/tenant_account_v2/group_views.py @@ -26,6 +26,10 @@ GroupMembership, OrganizationGroup, ) +from tenant_account_v2.share_notifications import ( + MembershipAction, + notify_group_membership_changed, +) logger = logging.getLogger(__name__) @@ -155,6 +159,14 @@ def members(self, request: Request, pk: str | None = None) -> Response: [GroupMembership(group=group, user_id=uid) for uid in user_ids_to_add], ignore_conflicts=True, ) + # The serializer already subtracts existing members, so nobody gets a + # second "you've been added" mail for a group they were already in. + notify_group_membership_changed( + group=group, + action=MembershipAction.ADDED, + user_ids=user_ids_to_add, + actor=request.user, + ) return Response( {"added_user_ids": user_ids_to_add}, status=status.HTTP_201_CREATED, @@ -178,6 +190,12 @@ def remove_member( deleted, _ = group.memberships.filter(user_id=user_id_int).delete() if not deleted: raise NotFound("User is not a member of this group.") + notify_group_membership_changed( + group=group, + action=MembershipAction.REMOVED, + user_ids=[user_id_int], + actor=request.user, + ) return Response(status=status.HTTP_204_NO_CONTENT) # --- resources shared with this group ------------------------------------ diff --git a/backend/tenant_account_v2/internal_urls.py b/backend/tenant_account_v2/internal_urls.py new file mode 100644 index 0000000000..4049c761a5 --- /dev/null +++ b/backend/tenant_account_v2/internal_urls.py @@ -0,0 +1,20 @@ +"""Internal API URLs for group-sharing email notifications.""" + +from django.urls import path + +from . import internal_views + +app_name = "group_notification_internal" + +urlpatterns = [ + path( + "resource-shared/", + internal_views.ResourceSharedWithGroupView.as_view(), + name="resource-shared", + ), + path( + "membership-changed/", + internal_views.GroupMembershipChangedView.as_view(), + name="membership-changed", + ), +] diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py new file mode 100644 index 0000000000..fb707df14e --- /dev/null +++ b/backend/tenant_account_v2/internal_views.py @@ -0,0 +1,91 @@ +"""Internal API views for group-sharing email notifications (UN-3494 / UNS-848). + +Mounted under ``/internal/`` and gated by ``InternalAPIAuthMiddleware``. The +notification worker calls these because ``workers/`` has no Django and every +step of the send — group expansion, org re-validation, resource lookup, the +email plugin — needs it. + +Failure contract: **any** unhandled problem must surface as non-2xx so the +queue redelivers. The one deliberate exception is a resource that no longer +exists, which returns 200 — retrying that can only fail again. +""" + +import logging + +from account_v2.models import Organization +from rest_framework import serializers, status +from rest_framework.exceptions import ValidationError +from rest_framework.request import Request +from rest_framework.response import Response +from rest_framework.views import APIView +from utils.user_context import UserContext + +from tenant_account_v2.group_notification_service import ( + ResourceNotFoundError, + send_membership_changed, + send_resource_shared, +) +from tenant_account_v2.share_notifications import MembershipAction + +logger = logging.getLogger(__name__) + + +class ResourceSharedWithGroupSerializer(serializers.Serializer): + """Payload of ``notify_resource_shared_with_group``.""" + + group_ids = serializers.ListField(child=serializers.IntegerField(), allow_empty=False) + actor_id = serializers.IntegerField() + resource_kind = serializers.CharField() + resource_id = serializers.CharField() + + +class GroupMembershipChangedSerializer(serializers.Serializer): + """Payload of ``notify_group_membership_changed``.""" + + group_id = serializers.IntegerField() + actor_id = serializers.IntegerField() + membership_action = serializers.ChoiceField( + choices=[a.value for a in MembershipAction] + ) + user_ids = serializers.ListField(child=serializers.IntegerField(), allow_empty=False) + + +class _GroupNotificationView(APIView): + """Shared org resolution for the group-notification endpoints.""" + + @staticmethod + def _organization() -> Organization: + organization = UserContext.get_organization() + if organization is None: + raise ValidationError( + "Organization context missing. Worker must send X-Organization-ID." + ) + return organization + + +class ResourceSharedWithGroupView(_GroupNotificationView): + """Mail every current member of the groups a resource was just shared with.""" + + def post(self, request: Request) -> Response: + serializer = ResourceSharedWithGroupSerializer(data=request.data) + serializer.is_valid(raise_exception=True) + data = serializer.validated_data + try: + send_resource_shared(organization=self._organization(), **data) + except ResourceNotFoundError as exc: + # Deleted between the share and the send — a retry cannot help. + logger.info("group-notification: dropping resource share (%s)", exc) + return Response({"status": "skipped"}, status=status.HTTP_200_OK) + return Response({"status": "success"}, status=status.HTTP_200_OK) + + +class GroupMembershipChangedView(_GroupNotificationView): + """Mail the users whose group membership just changed.""" + + def post(self, request: Request) -> Response: + serializer = GroupMembershipChangedSerializer(data=request.data) + serializer.is_valid(raise_exception=True) + send_membership_changed( + organization=self._organization(), **serializer.validated_data + ) + return Response({"status": "success"}, status=status.HTTP_200_OK) diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py new file mode 100644 index 0000000000..c4191a8a80 --- /dev/null +++ b/backend/tenant_account_v2/share_notifications.py @@ -0,0 +1,210 @@ +"""Enqueue hooks for group-sharing email notifications (UN-3494 / mfbt UNS-848). + +Two events earn a group's members an email: a resource shared with the group, +and a user added to or removed from it. Both are dispatched asynchronously — +the caller's request returns as soon as the write lands. + +The sending itself runs in ``workers/``, which is Django-free, so the worker +task is a thin HTTP shim back to :mod:`tenant_account_v2.internal_views`; the +backend does the ORM and plugin work. Transport is resolved per-org by the same +``resolve_transport`` gate the execution path uses — the PG queue where that is +enabled, Celery otherwise. + +The whole feature sits behind its own Flipt flag and fails closed everywhere: a +blind Flipt, a missing org, or any dispatch error means no notification, never +a broken share. +""" + +from __future__ import annotations + +import logging +import os +from collections.abc import Iterable +from enum import StrEnum +from typing import TYPE_CHECKING, Any + +from tenant_account_v2.shareable_resources import kind_for_instance +from unstract.core.data_models import is_pg_transport +from unstract.flags.feature_flag import check_feature_flag_status + +if TYPE_CHECKING: + from account_v2.models import User + + from tenant_account_v2.models import OrganizationGroup + +logger = logging.getLogger(__name__) + +# Rollout flag for the whole feature. Sibling of ``pg_queue.flags`` — kept in +# one place so a grep on the constant finds every gate. +GROUP_NOTIFICATION_FLAG_KEY = "group_sharing_notifications_enabled" + +NOTIFY_RESOURCE_SHARED_TASK = "notify_resource_shared_with_group" +NOTIFY_MEMBERSHIP_CHANGED_TASK = "notify_group_membership_changed" + +# Mirrors the workers' ``QueueName.NOTIFICATION`` — a local literal so the +# backend does not import the workers package (same as ``pipeline_dispatch``). +NOTIFICATION_QUEUE = "notifications" + + +class MembershipAction(StrEnum): + """What happened to a user's membership of a group.""" + + ADDED = "added" + REMOVED = "removed" + + +def notify_resource_shared_with_group( + *, resource: Any, groups: Iterable[OrganizationGroup], actor: User +) -> None: + """Queue "a resource was shared with your group" mail for newly added groups. + + Recipients are resolved at delivery time rather than frozen here: anyone + who leaves the org between the click and the send simply isn't in the fresh + lookup, so offboarding safety costs nothing. + """ + group_ids = sorted(group.pk for group in groups) + if not group_ids: + return + organization_id = _organization_slug(resource) + kind = kind_for_instance(resource) + if not organization_id or kind is None or not _feature_enabled(organization_id): + return + _dispatch_quietly( + task_name=NOTIFY_RESOURCE_SHARED_TASK, + kwargs={ + "group_ids": group_ids, + "actor_id": actor.pk, + "resource_kind": kind, + "resource_id": str(resource.pk), + "organization_id": organization_id, + }, + organization_id=organization_id, + entity_id=str(resource.pk), + ) + + +def notify_group_membership_changed( + *, + group: OrganizationGroup, + action: MembershipAction, + user_ids: Iterable[int], + actor: User, +) -> None: + """Queue "you were added to / removed from a group" mail for those users. + + Unlike a resource share, the user ids ride in the payload: on removal the + membership rows are already gone by delivery time, and on add a fresh group + lookup would mail every existing member too. + """ + recipients = sorted(user_ids) + if not recipients: + return + organization_id = _organization_slug(group) + if not organization_id or not _feature_enabled(organization_id): + return + _dispatch_quietly( + task_name=NOTIFY_MEMBERSHIP_CHANGED_TASK, + kwargs={ + "group_id": group.pk, + "actor_id": actor.pk, + "membership_action": str(action), + "user_ids": recipients, + "organization_id": organization_id, + }, + organization_id=organization_id, + entity_id=str(group.pk), + ) + + +def _feature_enabled(organization_id: str) -> bool: + """Whether group-sharing notifications are on for this org. Fails closed.""" + # Parse exactly as FliptClient does (``.lower()``, no ``.strip()``) so the + # two can never disagree on a value like " true". + if os.environ.get("FLIPT_SERVICE_AVAILABLE", "false").lower() != "true": + return False + try: + return bool( + check_feature_flag_status( + flag_key=GROUP_NOTIFICATION_FLAG_KEY, + entity_id=organization_id, + context={"organization_id": organization_id}, + ) + ) + except Exception: + logger.warning( + "group-notification: Flipt evaluation failed for org %s; skipping", + organization_id, + exc_info=True, + ) + return False + + +def _organization_slug(obj: Any) -> str | None: + """The owning org's string identifier (``Organization.organization_id``). + + This is the ``X-Organization-ID`` value the worker echoes back, not the DB + pk, and it is what ``resolve_transport`` expects. + """ + organization = getattr(obj, "organization", None) + return getattr(organization, "organization_id", None) + + +def _dispatch_quietly( + *, + task_name: str, + kwargs: dict[str, Any], + organization_id: str, + entity_id: str, +) -> None: + """Dispatch on the resolved transport; never let a failure reach the caller. + + The share or membership change has already been committed by the time this + runs — losing its email is not a reason to fail the request the user made. + """ + try: + _dispatch( + task_name=task_name, + kwargs=kwargs, + organization_id=organization_id, + entity_id=entity_id, + ) + except Exception: + logger.exception( + "group-notification: failed to dispatch %s for org %s", + task_name, + organization_id, + ) + + +def _dispatch( + *, + task_name: str, + kwargs: dict[str, Any], + organization_id: str, + entity_id: str, +) -> None: + # Lazy imports — ``backend.celery_service`` and ``pg_queue`` are heavier + # than this leaf module and importing them at load time risks a cycle + # during Django app loading. + from pg_queue.producer import enqueue_task + from workflow_manager.workflow_v2.transport import resolve_transport + + from backend.celery_service import app as celery_app + + transport = resolve_transport(execution_id=entity_id, organization_id=organization_id) + if is_pg_transport(transport): + msg_id = enqueue_task( + task_name=task_name, + queue=NOTIFICATION_QUEUE, + kwargs=kwargs, + org_id=organization_id, + ) + logger.info( + "group-notification: %s enqueued on PG queue %r (msg_id=%s)", + task_name, + NOTIFICATION_QUEUE, + msg_id, + ) + return + celery_app.send_task(task_name, kwargs=kwargs, queue=NOTIFICATION_QUEUE) + logger.info("group-notification: %s dispatched on Celery", task_name) diff --git a/backend/tenant_account_v2/shareable_resources.py b/backend/tenant_account_v2/shareable_resources.py index f528e2959f..2d55e0de0d 100644 --- a/backend/tenant_account_v2/shareable_resources.py +++ b/backend/tenant_account_v2/shareable_resources.py @@ -9,6 +9,7 @@ """ from dataclasses import dataclass +from typing import Any @dataclass(frozen=True) @@ -49,3 +50,26 @@ class ShareableResource: "id", ), ) + + +def descriptor_for_kind(kind: str) -> ShareableResource | None: + """Look up a descriptor by its ``kind`` key.""" + return next((r for r in SHAREABLE_RESOURCES if r.kind == kind), None) + + +def kind_for_instance(instance: Any) -> str | None: + """Reverse lookup: the ``kind`` of a resource instance, ``None`` if unlisted. + + Matches on the model's app label + class name so callers holding an + instance (e.g. the share endpoint) don't hardcode a type check per + resource. + """ + meta = instance._meta + return next( + ( + r.kind + for r in SHAREABLE_RESOURCES + if r.app_label == meta.app_label and r.model_name == meta.object_name + ), + None, + ) diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index 9a8db90afe..d74dc2bf43 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -827,6 +827,42 @@ services: profiles: - pg-queue + # Notification consumer — webhook POSTs and the group-share emails (UN-3494). + # Without this, a notification enqueued on PG is durably stored and never run. + # Every task here is one short outbound HTTP call, so it stays light. + worker-pg-notification: + image: unstract/worker-unified:${VERSION} + container_name: unstract-worker-pg-notification + restart: unless-stopped + command: ["pg-queue-consumer"] + ports: + - "8101:8090" + env_file: + - ../workers/.env + - ./essentials.env + depends_on: + - db + - redis + environment: + - ENVIRONMENT=development + - APPLICATION_NAME=unstract-worker-pg-notification + - WORKER_BARRIER_BACKEND=pg + - WORKER_PG_QUEUE_CONSUMER_WORKER_TYPE=notification + - WORKER_PG_QUEUE_CONSUMER_QUEUE=notifications + - WORKER_PG_QUEUE_CONSUMER_HEALTH_PORT=8090 + - WORKER_PG_QUEUE_CONSUMER_CONCURRENCY=${PG_NOTIFICATION_CONCURRENCY:-4} + # One internal-API call (30s timeout) plus a SendGrid batch for the + # largest group, with headroom. Health-stale sits at or above it. + - WORKER_PG_QUEUE_CONSUMER_VT_SECONDS=${PG_NOTIFICATION_VT_SECONDS:-120} + - WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS=${PG_NOTIFICATION_HEALTH_STALE_SECONDS:-180} + labels: + - traefik.enable=false + volumes: + - ./workflow_data:/data + - ${TOOL_REGISTRY_CONFIG_SRC_PATH}:/data/tool_registry_config + profiles: + - pg-queue + # Reaper / orchestrator — leader-elected loop. Run exactly ONE instance (it # elects a single leader via pg_orchestrator_lock; extra replicas idle as # standby). Besides barrier-orphan recovery it runs the PG scheduler tick diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 41d76f5cf9..f6e3d1ee9c 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -6,6 +6,7 @@ """ import os +import time from typing import Any import httpx @@ -467,6 +468,104 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An return process_notification(notification_type, priority=True, **kwargs) +# Retries for a transient backend problem (restart, 5xx). Kept inside the task +# because only the PG transport redelivers a failed message — on Celery a raise +# is terminal, so without this a rolling deploy would silently drop the email. +_GROUP_NOTIFICATION_ATTEMPTS = 3 +_GROUP_NOTIFICATION_RETRY_DELAY = 2.0 + + +def _post_group_notification(endpoint: str, organization_id: str, payload: dict) -> None: + """POST a group-notification job to the backend and insist it succeeded. + + Unlike ``_mark_buffer_outcome`` this deliberately **raises** on failure: + there is no reaper behind these rows, so a swallowed error would be a + silently unsent email. On the PG transport the raise also leaves the + message on the queue for redelivery, bounded by the consumer's attempt cap. + + A 4xx is not retried — a rejected payload will be rejected again. + """ + base_url = os.getenv("INTERNAL_API_BASE_URL") + api_key = os.getenv("INTERNAL_SERVICE_API_KEY") + if not base_url or not api_key: + raise RuntimeError( + "INTERNAL_API_BASE_URL / INTERNAL_SERVICE_API_KEY not set; " + "cannot send group notification" + ) + url = f"{base_url.rstrip('/')}/v1/group-notification/{endpoint}/" + headers = { + "Authorization": f"Bearer {api_key}", + # The backend resolves the tenant from this header; without it every + # org-scoped query comes back empty. + "X-Organization-ID": organization_id, + } + last_error = "" + for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): + try: + with httpx.Client(transport=httpx.HTTPTransport(retries=2)) as client: + response = client.post(url, headers=headers, json=payload, timeout=30.0) + except Exception as e: # noqa: BLE001 - transport failure, retry below + last_error = f"exception={e!r}" + else: + if response.status_code == 200: + return + last_error = f"http_{response.status_code} body={response.text[:200]}" + if response.status_code < 500: + break + if attempt < _GROUP_NOTIFICATION_ATTEMPTS: + logger.warning( + "Group notification %s attempt %d/%d failed (%s); retrying", + endpoint, + attempt, + _GROUP_NOTIFICATION_ATTEMPTS, + last_error, + ) + time.sleep(_GROUP_NOTIFICATION_RETRY_DELAY) + raise RuntimeError(f"Group notification {endpoint} failed: {last_error}") + + +@worker_task(name="notify_resource_shared_with_group") +def notify_resource_shared_with_group( + group_ids: list[int], + actor_id: int, + resource_kind: str, + resource_id: str, + organization_id: str, +) -> None: + """Email every current member of the groups a resource was shared with.""" + _post_group_notification( + "resource-shared", + organization_id, + { + "group_ids": group_ids, + "actor_id": actor_id, + "resource_kind": resource_kind, + "resource_id": resource_id, + }, + ) + + +@worker_task(name="notify_group_membership_changed") +def notify_group_membership_changed( + group_id: int, + actor_id: int, + membership_action: str, + user_ids: list[int], + organization_id: str, +) -> None: + """Email the users whose membership of a group just changed.""" + _post_group_notification( + "membership-changed", + organization_id, + { + "group_id": group_id, + "actor_id": actor_id, + "membership_action": membership_action, + "user_ids": user_ids, + }, + ) + + @worker_task(name="notification_health_check") def notification_health_check() -> dict[str, Any]: """Health check task for notification worker.""" From d8b1008b70fa2010e28a772d0f2d71922cccfdc0 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Mon, 3 Aug 2026 19:26:01 +0530 Subject: [PATCH 02/28] UN-3494 [FIX] Restore direct-user sharing emails on the share endpoint UN-2977 moved sharing from PATCH to POST /{id}/share/, but the mixin's share action only diffed the groups axis. The per-viewset _notify_shared_users hooks stayed on partial_update, which nothing calls anymore, so sharing a resource with a user sent no email. Snapshot every declared axis and invoke the hook after the commit; declare it on the mixin as a no-op for hosts without a direct-share email. Co-Authored-By: Claude Opus 5 --- backend/permissions/resource_share_views.py | 29 ++++++++++++++------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index f688569115..71030b0a7f 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -98,24 +98,36 @@ def share(self, request: Request, pk: str | None = None) -> Response: resource = self.get_object() # type: ignore[attr-defined] desired = _extract_desired_share_state(request.data) - # Only the groups axis is diffed: it is the one that notifies, and - # snapshotting ``shared_users`` too would fetch every viewer twice for - # nothing. Reads go through ``ResourceGroupShare``, so no refresh is - # needed between the two. - groups_before = self._read_axis(resource, "shared_groups") + before = self.snapshot_share_axes(resource) ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) # ``_commit`` is the only atomic block on this path, so it has already - # committed — the diff reads persisted state and can never announce a + # committed — the diffs read persisted state and can never announce a # share that rolled back. notify_resource_shared_with_group( resource=resource, - groups=self._read_axis(resource, "shared_groups") - groups_before, + # ``.get`` — lookups narrows ``share_axes`` to users only. + groups=self._read_axis(resource, "shared_groups") + - before.get("shared_groups", set()), actor=request.user, ) + self._notify_shared_users(resource, before, request.data, request.user) return Response(status=status.HTTP_200_OK) + def _notify_shared_users( + self, + instance: Any, + before: dict[str, set[Any]], + request_data: dict[str, Any], + actor: Any, + /, + ) -> None: + """Email users newly added to ``shared_users``. + + Positional-only: hosts override with their own resource name and type. + """ + @action(detail=True, methods=["get"], url_path="effective-members") def effective_members(self, request: Request, pk: str | None = None) -> Response: """Return all users with access (direct/group/org), priority-deduped.""" @@ -132,8 +144,7 @@ def effective_members(self, request: Request, pk: str | None = None) -> Response def snapshot_share_axes(self, instance: Model) -> dict[str, set[Any]]: """Capture every declared axis's current contents. - Call BEFORE ``super().partial_update(...)``; pair with - :meth:`diff_share_axes` afterward. + Call BEFORE the write; pair with :meth:`diff_share_axes` afterward. """ return {axis: self._read_axis(instance, axis) for axis in self.share_axes} From 77ca124b5958f6ec34d5f9e68fe0af2bc1c78bee Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 4 Aug 2026 19:00:03 +0530 Subject: [PATCH 03/28] UN-3494 [FEAT] Email users and group members when resource access is revoked MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sharing already emailed on grant; revoking told nobody. Both axes now notify, and the seven duplicated copies of the user hook collapse into the share mixin. - ResourceShareManagementMixin gains a concrete _notify_shared_users covering grant and revoke, driven by the OwnerManagementMixin seam every host already declares. The seven per-viewset overrides and their dead partial_update wrappers go with it — a host override would otherwise shadow the mixin and silently swallow the revoke mail. - share() diffs both axes through _read_axis directly; AxisDiff, snapshot_share_axes, diff_share_axes and the share_axes ClassVar had no callers left. - Group revoke rides the existing resource-shared route with a share_action discriminator, mirroring membership-changed — no new endpoint or worker task. Defaulted at every hop so in-flight messages still run. - Suppressed when the user still reaches the resource via a group or shared_to_org: losing one axis is not losing access. Co-Authored-By: Claude Opus 5 --- backend/adapter_processor_v2/views.py | 47 ----- backend/api_v2/api_deployment_views.py | 37 ---- backend/connector_v2/views.py | 43 ----- backend/permissions/resource_share_views.py | 171 +++++++++++------- backend/pipeline_v2/views.py | 54 ------ .../prompt_studio_core_v2/views.py | 48 ----- .../group_notification_service.py | 11 +- backend/tenant_account_v2/internal_views.py | 8 +- .../tenant_account_v2/share_notifications.py | 45 ++++- backend/workflow_manager/workflow_v2/views.py | 45 ----- workers/notification/tasks.py | 7 +- 11 files changed, 161 insertions(+), 355 deletions(-) diff --git a/backend/adapter_processor_v2/views.py b/backend/adapter_processor_v2/views.py index f2aef82b0c..22f75d7b17 100644 --- a/backend/adapter_processor_v2/views.py +++ b/backend/adapter_processor_v2/views.py @@ -404,17 +404,6 @@ def destroy( raise DeleteAdapterInUseError(adapter_name=adapter_instance.adapter_name) return Response(status=status.HTTP_204_NO_CONTENT) - def partial_update( - self, request: Request, *args: tuple[Any], **kwargs: dict[str, Any] - ) -> Response: - adapter = self.get_object() - before = self.snapshot_share_axes(adapter) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200 and notification_plugin: - self._notify_shared_users(adapter, before, request.data, request.user) - return response - @action(detail=True, methods=["post"], url_path="share") def share(self, request: Request, pk: str | None = None) -> Response: """Apply share state, then clear default-adapter links for any user @@ -461,42 +450,6 @@ def on_owner_removed(self, resource: AdapterInstance, user: User) -> None: return self._clear_default_adapter_for_removed_users(resource, {user.pk}) - def _notify_shared_users( - self, - adapter: AdapterInstance, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort).""" - users_diff = self.diff_share_axes(adapter, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - try: - adapter_type_to_resource = { - "LLM": ResourceType.LLM.value, - "EMBEDDING": ResourceType.EMBEDDING.value, - "VECTOR_DB": ResourceType.VECTOR_DB.value, - "X2TEXT": ResourceType.X2TEXT.value, - } - resource_type = adapter_type_to_resource.get( - adapter.adapter_type, ResourceType.LLM.value - ) - service_class = notification_plugin["service_class"] - notification_service = service_class() - notification_service.send_sharing_notification( - resource_type=resource_type, - resource_name=adapter.adapter_name, - resource_id=str(adapter.id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=adapter, - ) - except Exception as e: - logger.exception("Failed to send sharing notification: %s", e) - def _clear_default_adapter_for_removed_users( self, adapter: AdapterInstance, diff --git a/backend/api_v2/api_deployment_views.py b/backend/api_v2/api_deployment_views.py index 8b96bcd67e..f9106e312f 100644 --- a/backend/api_v2/api_deployment_views.py +++ b/backend/api_v2/api_deployment_views.py @@ -414,40 +414,3 @@ def list_of_shared_users(self, request: Request, pk: str | None = None) -> Respo instance = self.get_object() serializer = SharedUserListSerializer(instance) return Response(serializer.data) - - def partial_update(self, request: Request, *args: Any, **kwargs: Any) -> Response: - """Override partial_update to handle sharing notifications.""" - instance = self.get_object() - before = self.snapshot_share_axes(instance) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200 and notification_plugin: - self._notify_shared_users(instance, before, request.data, request.user) - return response - - def _notify_shared_users( - self, - instance: APIDeployment, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort).""" - users_diff = self.diff_share_axes(instance, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - try: - service_class = notification_plugin["service_class"] - notification_service = service_class() - notification_service.send_sharing_notification( - resource_type=ResourceType.API_DEPLOYMENT.value, - resource_name=instance.display_name, - resource_id=str(instance.id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=instance, - ) - except Exception as e: - logger.exception("Failed to send sharing notification: %s", e) diff --git a/backend/connector_v2/views.py b/backend/connector_v2/views.py index fd75b749db..c3bb018ff1 100644 --- a/backend/connector_v2/views.py +++ b/backend/connector_v2/views.py @@ -36,7 +36,6 @@ notification_plugin = get_plugin("notification") if notification_plugin: from plugins.notification.constants import ResourceType - from plugins.notification.sharing_notification import SharingNotificationService logger = logging.getLogger(__name__) @@ -286,45 +285,3 @@ def perform_destroy(self, instance: ConnectorInstance) -> None: f" named {instance.connector_name}" ) raise DeleteConnectorInUseError(connector_name=instance.connector_name) - - def partial_update(self, request: Request, *args: Any, **kwargs: Any) -> Response: - """Override to handle sharing notifications.""" - instance = self.get_object() - before = self.snapshot_share_axes(instance) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200 and notification_plugin: - self._notify_shared_users(instance, before, request.data, request.user) - return response - - def _notify_shared_users( - self, - instance: ConnectorInstance, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort).""" - users_diff = self.diff_share_axes(instance, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - try: - SharingNotificationService().send_sharing_notification( - resource_type=ResourceType.CONNECTOR.value, - resource_name=instance.connector_name, - resource_id=str(instance.id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=instance, - ) - logger.info( - "Sent sharing notifications for connector to %d users", - len(users_diff.added), - ) - except Exception as e: - logger.exception( - "Failed to send sharing notification, continuing update though: %s", - str(e), - ) diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 71030b0a7f..3b13cdc988 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -1,22 +1,26 @@ """Shared share-management surface for resource ViewSets. -The mixin is **axis-agnostic** — it operates over the sharing "axes" declared -in :attr:`ResourceShareManagementMixin.share_axes`. ``shared_users`` is an M2M -on the resource model, while ``shared_groups`` is stored polymorphically in -``ResourceGroupShare`` (not an M2M) and routed through the sharing helpers; new -axes can be added by extending that attribute. +The mixin is **axis-agnostic** — it reads the sharing "axes" named in +``_SUPPORTED_SHARE_AXES``. ``shared_users`` is the direct-viewer axis, backed by +VIEWER membership rows, while ``shared_groups`` is stored polymorphically in +``ResourceGroupShare`` (not an M2M) and routed through the sharing helpers. """ -from dataclasses import dataclass, field -from typing import Any, ClassVar +import logging +from typing import Any from django.db.models import Model +from plugins import get_plugin from rest_framework import status from rest_framework.decorators import action from rest_framework.exceptions import ValidationError from rest_framework.request import Request from rest_framework.response import Response +logger = logging.getLogger(__name__) + +notification_plugin = get_plugin("notification") + _SUPPORTED_SHARE_AXES = ("shared_users", "shared_groups", "shared_to_org") @@ -55,30 +59,78 @@ def _coerce_id_list(axis: str, value: Any) -> list[int]: return coerced -@dataclass -class AxisDiff: - """Pre/post snapshot for a single share axis (M2M field).""" - - before: set[Any] = field(default_factory=set) - after: set[Any] = field(default_factory=set) +def _notification_context(view: Any, instance: Any) -> tuple[str, str] | None: + """Resolve ``(resource_type, resource_name)`` for the email senders. - @property - def added(self) -> set[Any]: - return self.after - self.before - - @property - def removed(self) -> set[Any]: - return self.before - self.after + ``None`` when the plugin is absent or the host ViewSet has not opted in by + setting ``notification_resource_name_field`` and overriding + ``get_notification_resource_type`` (both declared on + ``OwnerManagementMixin``, which every share host also mixes in). + """ + name_field = getattr(view, "notification_resource_name_field", None) + resolve_type = getattr(view, "get_notification_resource_type", None) + if not notification_plugin or not name_field or resolve_type is None: + return None + resource_type = resolve_type(instance) + resource_name = getattr(instance, name_field, None) + if resource_type is None or not resource_name: + return None + return resource_type, resource_name + + +def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: + """Narrow ``users`` to those with no remaining access to ``instance``. + + Someone dropped from ``shared_users`` may still reach the resource via a + group or an org-wide share; telling them their access was removed would be + wrong. + """ + if not users: + return [] + from tenant_account_v2.sharing_helpers import compute_effective_members + + retained = {member["user_id"] for member in compute_effective_members(instance)} + return [user for user in users if user.pk not in retained] + + +def _send_share_notification( + instance: Model, context: tuple[str, str], users: set[Any], actor: Any +) -> None: + """Email users newly granted direct access. Best-effort.""" + resource_type, resource_name = context + try: + notification_plugin["service_class"]().send_sharing_notification( + resource_type=resource_type, + resource_name=resource_name, + resource_id=str(instance.pk), + shared_by=actor, + shared_to=list(users), + resource_instance=instance, + ) + except Exception: + logger.exception("Failed to send sharing notification for %s", instance.pk) + + +def _send_revoke_notification( + instance: Model, context: tuple[str, str], users: list[Any], actor: Any +) -> None: + """Email users whose direct access was revoked. Best-effort.""" + resource_type, resource_name = context + try: + notification_plugin["service_class"]().send_access_removed_notification( + resource_type=resource_type, + resource_name=resource_name, + resource_id=str(instance.pk), + removed_from=users, + removed_by=actor, + resource_instance=instance, + ) + except Exception: + logger.exception("Failed to send access-removed notification for %s", instance.pk) class ResourceShareManagementMixin: - """Adds the shared share-management surface to a resource ViewSet. - - Subclasses declare share axes via :attr:`share_axes`. The default - covers ``shared_users`` + ``shared_groups``. - """ - - share_axes: ClassVar[tuple[str, ...]] = ("shared_users", "shared_groups") + """Adds the shared share-management surface to a resource ViewSet.""" @action(detail=True, methods=["post"], url_path="share") def share(self, request: Request, pk: str | None = None) -> Response: @@ -92,41 +144,55 @@ def share(self, request: Request, pk: str | None = None) -> Response: ``ShareAuthorizationService``. """ from tenant_account_v2.share_notifications import ( - notify_resource_shared_with_group, + notify_resource_group_share_changed, ) from tenant_account_v2.sharing_helpers import ShareAuthorizationService resource = self.get_object() # type: ignore[attr-defined] desired = _extract_desired_share_state(request.data) - before = self.snapshot_share_axes(resource) + users_before = self._read_axis(resource, "shared_users") + groups_before = self._read_axis(resource, "shared_groups") ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) # ``_commit`` is the only atomic block on this path, so it has already # committed — the diffs read persisted state and can never announce a # share that rolled back. - notify_resource_shared_with_group( + resource.refresh_from_db() + users_after = self._read_axis(resource, "shared_users") + groups_after = self._read_axis(resource, "shared_groups") + notify_resource_group_share_changed( resource=resource, - # ``.get`` — lookups narrows ``share_axes`` to users only. - groups=self._read_axis(resource, "shared_groups") - - before.get("shared_groups", set()), + added=groups_after - groups_before, + removed=groups_before - groups_after, actor=request.user, ) - self._notify_shared_users(resource, before, request.data, request.user) + self._notify_shared_users( + resource, users_after - users_before, users_before - users_after, request.user + ) return Response(status=status.HTTP_200_OK) def _notify_shared_users( self, instance: Any, - before: dict[str, set[Any]], - request_data: dict[str, Any], + added: set[Any], + removed: set[Any], actor: Any, /, ) -> None: - """Email users newly added to ``shared_users``. + """Email users granted or denied direct access. - Positional-only: hosts override with their own resource name and type. + Resource type and name come from the host's ``OwnerManagementMixin`` + seam, so every share host is covered without an override. """ + context = _notification_context(self, instance) + if context is None: + return + if added: + _send_share_notification(instance, context, added, actor) + revoked = _users_left_without_access(instance, removed) + if revoked: + _send_revoke_notification(instance, context, revoked, actor) @action(detail=True, methods=["get"], url_path="effective-members") def effective_members(self, request: Request, pk: str | None = None) -> Response: @@ -141,35 +207,6 @@ def effective_members(self, request: Request, pk: str | None = None) -> Response members = compute_effective_members(self.get_object()) # type: ignore[attr-defined] return Response(EffectiveMemberSerializer(members, many=True).data) - def snapshot_share_axes(self, instance: Model) -> dict[str, set[Any]]: - """Capture every declared axis's current contents. - - Call BEFORE the write; pair with :meth:`diff_share_axes` afterward. - """ - return {axis: self._read_axis(instance, axis) for axis in self.share_axes} - - def diff_share_axes( - self, - instance: Model, - before: dict[str, set[Any]], - request_data: dict[str, Any], - ) -> dict[str, AxisDiff]: - """Diff each axis that was touched by the request. - - Returns a dict keyed by axis name with only the axes present in - ``request_data`` — callers can skip notification fan-out for axes - the client did not modify. - """ - instance.refresh_from_db() - return { - axis: AxisDiff( - before=before[axis], - after=self._read_axis(instance, axis), - ) - for axis in self.share_axes - if axis in request_data - } - @staticmethod def _read_axis(instance: Model, axis: str) -> set[Any]: """Return the current set of related objects on the given axis. diff --git a/backend/pipeline_v2/views.py b/backend/pipeline_v2/views.py index ec7e7720f3..2683ac2bcb 100644 --- a/backend/pipeline_v2/views.py +++ b/backend/pipeline_v2/views.py @@ -187,60 +187,6 @@ def list_of_shared_users(self, request: Request, pk: str | None = None) -> Respo serializer = SharedUserListSerializer(pipeline) return Response(serializer.data, status=status.HTTP_200_OK) - def partial_update(self, request: Request, *args: Any, **kwargs: Any) -> Response: - """Override to handle sharing notifications.""" - instance = self.get_object() - before = self.snapshot_share_axes(instance) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200 and notification_plugin: - self._notify_shared_users(instance, before, request.data, request.user) - return response - - def _notify_shared_users( - self, - instance: Pipeline, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort). - - Only ETL/TASK pipelines map to a notification ``ResourceType``; - DEFAULT/APP pipelines have no analogue and skip the fan-out. - """ - users_diff = self.diff_share_axes(instance, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - if instance.pipeline_type not in ( - ResourceType.ETL.value, - ResourceType.TASK.value, - ): - return - try: - service_class = notification_plugin["service_class"] - notification_service = service_class() - notification_service.send_sharing_notification( - resource_type=instance.pipeline_type, - resource_name=instance.pipeline_name, - resource_id=str(instance.id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=instance, - ) - logger.info( - "Sent sharing notifications for %s to %d users", - instance.pipeline_type, - len(users_diff.added), - ) - except Exception as e: - logger.exception( - "Failed to send sharing notification, continuing update though: %s", - str(e), - ) - @action(detail=True, methods=["get"]) def download_postman_collection( self, request: Request, pk: str | None = None diff --git a/backend/prompt_studio/prompt_studio_core_v2/views.py b/backend/prompt_studio/prompt_studio_core_v2/views.py index cd65f2de77..6328293b0c 100644 --- a/backend/prompt_studio/prompt_studio_core_v2/views.py +++ b/backend/prompt_studio/prompt_studio_core_v2/views.py @@ -335,54 +335,6 @@ def destroy( ) return super().destroy(request, *args, **kwargs) - def partial_update( - self, request: Request, *args: tuple[Any], **kwargs: dict[str, Any] - ) -> Response: - custom_tool = self.get_object() - before = self.snapshot_share_axes(custom_tool) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200: - self._notify_shared_users(custom_tool, before, request.data, request.user) - return response - - def _notify_shared_users( - self, - custom_tool: CustomTool, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort).""" - notification_plugin = get_plugin("notification") - if not notification_plugin: - return - users_diff = self.diff_share_axes(custom_tool, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - - from plugins.notification.constants import ResourceType - - try: - service_class = notification_plugin["service_class"] - notification_service = service_class() - notification_service.send_sharing_notification( - resource_type=ResourceType.TEXT_EXTRACTOR.value, - resource_name=custom_tool.tool_name, - resource_id=str(custom_tool.tool_id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=custom_tool, - ) - except Exception as e: - logger.exception( - "Failed to send sharing notification for custom tool %s: %s", - custom_tool.tool_id, - str(e), - ) - @action(detail=True, methods=["get"]) def get_select_choices(self, request: HttpRequest) -> Response: """Method to return all static dropdown field values. diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 53f2bc7f82..ccdc3d9989 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -20,7 +20,7 @@ from plugins import get_plugin from tenant_account_v2.models import OrganizationGroup, OrganizationMember -from tenant_account_v2.share_notifications import MembershipAction +from tenant_account_v2.share_notifications import MembershipAction, ShareAction from tenant_account_v2.shareable_resources import ShareableResource, descriptor_for_kind if TYPE_CHECKING: @@ -62,11 +62,12 @@ def send_resource_shared( actor_id: int, resource_kind: str, resource_id: str, + share_action: str = ShareAction.SHARED.value, ) -> None: - """Mail every current member of each group that a resource was shared. + """Mail every current member of each group whose resource access changed. One email per group, so ``group_name`` in the template is always the group - the recipient actually belongs to. + the recipient actually belongs to. ``share_action`` picks the wording. """ service = _service() if service is None: @@ -90,9 +91,10 @@ def send_resource_shared( organization, group.memberships.values_list("user_id", flat=True) ) logger.info( - "group-notification: task=%s group_id=%s recipient_count=%d", + "group-notification: task=%s group_id=%s action=%s recipient_count=%d", "notify_resource_shared_with_group", group.pk, + share_action, len(recipients), ) if not recipients: @@ -105,6 +107,7 @@ def send_resource_shared( shared_by=actor, shared_to=recipients, resource_instance=resource, + share_action=ShareAction(share_action).value, ) diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index fb707df14e..0e959c9b5d 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -25,7 +25,7 @@ send_membership_changed, send_resource_shared, ) -from tenant_account_v2.share_notifications import MembershipAction +from tenant_account_v2.share_notifications import MembershipAction, ShareAction logger = logging.getLogger(__name__) @@ -37,6 +37,10 @@ class ResourceSharedWithGroupSerializer(serializers.Serializer): actor_id = serializers.IntegerField() resource_kind = serializers.CharField() resource_id = serializers.CharField() + # Defaulted so messages enqueued before this field existed still validate. + share_action = serializers.ChoiceField( + choices=[a.value for a in ShareAction], default=ShareAction.SHARED.value + ) class GroupMembershipChangedSerializer(serializers.Serializer): @@ -64,7 +68,7 @@ def _organization() -> Organization: class ResourceSharedWithGroupView(_GroupNotificationView): - """Mail every current member of the groups a resource was just shared with.""" + """Mail every current member of the groups whose resource access just changed.""" def post(self, request: Request) -> Response: serializer = ResourceSharedWithGroupSerializer(data=request.data) diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index c4191a8a80..526f926a04 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -1,8 +1,8 @@ """Enqueue hooks for group-sharing email notifications (UN-3494 / mfbt UNS-848). -Two events earn a group's members an email: a resource shared with the group, -and a user added to or removed from it. Both are dispatched asynchronously — -the caller's request returns as soon as the write lands. +Two events earn a group's members an email: a resource shared with or revoked +from the group, and a user added to or removed from it. Both are dispatched +asynchronously — the caller's request returns as soon as the write lands. The sending itself runs in ``workers/``, which is Django-free, so the worker task is a thin HTTP shim back to :mod:`tenant_account_v2.internal_views`; the @@ -53,14 +53,44 @@ class MembershipAction(StrEnum): REMOVED = "removed" -def notify_resource_shared_with_group( - *, resource: Any, groups: Iterable[OrganizationGroup], actor: User +class ShareAction(StrEnum): + """What happened to a group's access to a resource.""" + + SHARED = "shared" + REVOKED = "revoked" + + +def notify_resource_group_share_changed( + *, + resource: Any, + added: Iterable[OrganizationGroup], + removed: Iterable[OrganizationGroup], + actor: User, +) -> None: + """Queue group mail for a resource just shared with / revoked from groups.""" + for share_action, groups in ( + (ShareAction.SHARED, added), + (ShareAction.REVOKED, removed), + ): + _notify_group_share( + resource=resource, groups=groups, share_action=share_action, actor=actor + ) + + +def _notify_group_share( + *, + resource: Any, + groups: Iterable[OrganizationGroup], + share_action: ShareAction, + actor: User, ) -> None: - """Queue "a resource was shared with your group" mail for newly added groups. + """Queue one group-share event. Recipients are resolved at delivery time rather than frozen here: anyone who leaves the org between the click and the send simply isn't in the fresh - lookup, so offboarding safety costs nothing. + lookup, so offboarding safety costs nothing. Unlike a membership removal, + revoking a group's access leaves the group and its members intact, so the + fresh lookup still finds everyone who needs telling. """ group_ids = sorted(group.pk for group in groups) if not group_ids: @@ -76,6 +106,7 @@ def notify_resource_shared_with_group( "actor_id": actor.pk, "resource_kind": kind, "resource_id": str(resource.pk), + "share_action": str(share_action), "organization_id": organization_id, }, organization_id=organization_id, diff --git a/backend/workflow_manager/workflow_v2/views.py b/backend/workflow_manager/workflow_v2/views.py index fefba8c21a..e567f39e2b 100644 --- a/backend/workflow_manager/workflow_v2/views.py +++ b/backend/workflow_manager/workflow_v2/views.py @@ -173,51 +173,6 @@ def perform_create(self, serializer: WorkflowSerializer) -> Workflow: raise WorkflowGenerationError return workflow - def partial_update(self, request: Request, *args: Any, **kwargs: Any) -> Response: - """Override partial_update to handle sharing notifications.""" - workflow = self.get_object() - before = self.snapshot_share_axes(workflow) - - response = super().partial_update(request, *args, **kwargs) - if response.status_code == 200 and notification_plugin: - self._notify_shared_users(workflow, before, request.data, request.user) - return response - - def _notify_shared_users( - self, - workflow: Workflow, - before: dict[str, set[Any]], - request_data: dict[str, Any], - actor: Any, - ) -> None: - """Email users newly added to ``shared_users`` (best-effort).""" - users_diff = self.diff_share_axes(workflow, before, request_data).get( - "shared_users" - ) - if not (users_diff and users_diff.added): - return - try: - service_class = notification_plugin["service_class"] - notification_service = service_class() - notification_service.send_sharing_notification( - resource_type=ResourceType.WORKFLOW.value, - resource_name=workflow.workflow_name, - resource_id=str(workflow.id), - shared_by=actor, - shared_to=list(users_diff.added), - resource_instance=workflow, - ) - logger.info( - "Sent sharing notifications for workflow %s to %d users", - workflow.id, - len(users_diff.added), - ) - except Exception as e: - logger.exception( - "Failed to send sharing notification, continuing update though: %s", - str(e), - ) - def get_execution(self, request: Request, pk: str) -> Response: execution = WorkflowHelper.get_current_execution(pk) return Response(make_execution_response(execution), status=status.HTTP_200_OK) diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index f6e3d1ee9c..10945e34c8 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -531,8 +531,12 @@ def notify_resource_shared_with_group( resource_kind: str, resource_id: str, organization_id: str, + share_action: str = "shared", ) -> None: - """Email every current member of the groups a resource was shared with.""" + """Email every current member of the groups whose access just changed. + + ``share_action`` defaults so messages enqueued before it existed still run. + """ _post_group_notification( "resource-shared", organization_id, @@ -541,6 +545,7 @@ def notify_resource_shared_with_group( "actor_id": actor_id, "resource_kind": resource_kind, "resource_id": resource_id, + "share_action": share_action, }, ) From 3392ebea55fa9cb21c5a026d12c27f5a86cfd239 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 13:41:22 +0530 Subject: [PATCH 04/28] UN-3494 [FIX] Gate co-owner removal behind the share modal's Apply button MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adding a co-owner was staged until Apply, but revoking one fired the DELETE straight from the Popconfirm — so Cancel could not undo it, Apply stayed disabled for a removal-only edit, and the revoke email went out on click. Stage the roster the way SharePermission does: one selected-owners list seeded from the server, edited locally by both add and revoke, committed only by Apply. Collapse the hook's two mutation callbacks into one onApplyCoOwners that runs adds before removes (so a one-shot owner swap clears the backend's last-owner guard), refreshes once, and emits one summary alert. Apply now closes on a clean run and stays open on failure, matching useShareModal. Co-Authored-By: Claude Opus 5 --- .../api-deployment/ApiDeployment.jsx | 7 +- .../pipelines/Pipelines.jsx | 7 +- .../co-owner-management/CoOwnerManagement.css | 4 - .../co-owner-management/CoOwnerManagement.jsx | 218 ++++++++---------- .../co-owner-management/CoOwnerModal.jsx | 4 +- frontend/src/hooks/useCoOwnerManagement.jsx | 139 ++++++----- 6 files changed, 171 insertions(+), 208 deletions(-) diff --git a/frontend/src/components/deployments/api-deployment/ApiDeployment.jsx b/frontend/src/components/deployments/api-deployment/ApiDeployment.jsx index b1e5bd0708..4f2b092975 100644 --- a/frontend/src/components/deployments/api-deployment/ApiDeployment.jsx +++ b/frontend/src/components/deployments/api-deployment/ApiDeployment.jsx @@ -65,8 +65,7 @@ function ApiDeployment() { coOwnerAllUsers, coOwnerResourceId, handleCoOwner: handleCoOwnerAction, - onAddCoOwner, - onRemoveCoOwner, + onApplyCoOwners, } = useCoOwnerManagement({ service: apiDeploymentsApiService, setAlertDetails, @@ -408,10 +407,8 @@ function ApiDeployment() { resourceType="API Deployment" allUsers={coOwnerAllUsers} coOwners={coOwnerData.coOwners} - createdBy={coOwnerData.createdBy} loading={coOwnerLoading} - onAddCoOwner={onAddCoOwner} - onRemoveCoOwner={onRemoveCoOwner} + onApplyCoOwners={onApplyCoOwners} /> ); diff --git a/frontend/src/components/pipelines-or-deployments/pipelines/Pipelines.jsx b/frontend/src/components/pipelines-or-deployments/pipelines/Pipelines.jsx index edde00911e..52c38e0f8f 100644 --- a/frontend/src/components/pipelines-or-deployments/pipelines/Pipelines.jsx +++ b/frontend/src/components/pipelines-or-deployments/pipelines/Pipelines.jsx @@ -69,8 +69,7 @@ function Pipelines({ type }) { coOwnerAllUsers, coOwnerResourceId, handleCoOwner: handleCoOwnerAction, - onAddCoOwner, - onRemoveCoOwner, + onApplyCoOwners, } = useCoOwnerManagement({ service: pipelineApiService, setAlertDetails, @@ -487,10 +486,8 @@ function Pipelines({ type }) { resourceType="Pipeline" allUsers={coOwnerAllUsers} coOwners={coOwnerData.coOwners} - createdBy={coOwnerData.createdBy} loading={coOwnerLoading} - onAddCoOwner={onAddCoOwner} - onRemoveCoOwner={onRemoveCoOwner} + onApplyCoOwners={onApplyCoOwners} /> )} diff --git a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.css b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.css index c7698eea62..75eb050515 100644 --- a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.css +++ b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.css @@ -3,10 +3,6 @@ margin-bottom: 16px; } -.co-owner-creator-tag { - margin-left: 8px; -} - .co-owner-modal .shared-user-avatar { background-color: #00a6ed; margin-right: 15px; diff --git a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx index 7d8648e13a..bdc58ed206 100644 --- a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx +++ b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx @@ -13,7 +13,7 @@ import { Typography, } from "antd"; import PropTypes from "prop-types"; -import { useMemo, useState } from "react"; +import { useEffect, useMemo, useState } from "react"; import { SpinnerLoader } from "../spinner-loader/SpinnerLoader"; import "./CoOwnerManagement.css"; @@ -26,82 +26,84 @@ function CoOwnerManagement({ allUsers, coOwners, loading, - onAddCoOwner, - onRemoveCoOwner, + onApplyCoOwners, }) { - const [pendingAdds, setPendingAdds] = useState([]); - const [removingUserId, setRemovingUserId] = useState(null); + // Staged roster. Adds and removals both edit this list only — nothing reaches + // the API until Apply, the same contract as the share modal. + const [selectedOwners, setSelectedOwners] = useState([]); const [applying, setApplying] = useState(false); - const ownersList = coOwners || []; - const totalOwners = ownersList.length; - - // Exclude both existing co-owners and pending adds from dropdown - const availableUsers = useMemo(() => { - const coOwnerIds = new Set((coOwners || []).map((u) => u?.id?.toString())); - const pendingIds = new Set(pendingAdds.map((u) => u?.id?.toString())); - return (allUsers || []).filter( - (user) => - !coOwnerIds.has(user?.id?.toString()) && - !pendingIds.has(user?.id?.toString()), - ); - }, [allUsers, coOwners, pendingAdds]); + const ownersList = useMemo(() => coOwners || [], [coOwners]); + + // Re-seed whenever the server roster changes: on open, on resource switch, and + // after an Apply. Doubles as the reset — most hosts leave this modal mounted, + // and the hook can close it without ``handleCancel`` (404 / fetch-error), so + // staged edits must not leak into the next resource. + useEffect(() => { + setSelectedOwners(ownersList); + }, [ownersList]); + + const selectedIds = useMemo( + () => new Set(selectedOwners.map((u) => u?.id?.toString())), + [selectedOwners], + ); + + const availableUsers = useMemo( + () => (allUsers || []).filter((u) => !selectedIds.has(u?.id?.toString())), + [allUsers, selectedIds], + ); + + const { addUsers, removeUsers } = useMemo(() => { + const ownerIds = new Set(ownersList.map((u) => u?.id?.toString())); + return { + addUsers: selectedOwners.filter((u) => !ownerIds.has(u?.id?.toString())), + removeUsers: ownersList.filter( + (u) => !selectedIds.has(u?.id?.toString()), + ), + }; + }, [ownersList, selectedOwners, selectedIds]); + + const hasChanges = addUsers.length > 0 || removeUsers.length > 0; const handleSelect = (userId) => { const user = (allUsers || []).find( (u) => u?.id?.toString() === userId?.toString(), ); if (user) { - setPendingAdds((prev) => [...prev, user]); + setSelectedOwners((prev) => [...prev, user]); } }; - const handleRemovePending = (userId) => { - setPendingAdds((prev) => + const handleRemove = (userId) => { + setSelectedOwners((prev) => prev.filter((u) => u?.id?.toString() !== userId?.toString()), ); }; - const handleRemoveExisting = async (userId) => { - setRemovingUserId(userId); - try { - await onRemoveCoOwner(resourceId, userId); - } finally { - setRemovingUserId(null); - } - }; - const handleApply = async () => { - if (pendingAdds.length === 0) return; - const usersToAdd = [...pendingAdds]; + if (!hasChanges) { + return; + } setApplying(true); try { - const userIds = usersToAdd.map((user) => user.id); - await onAddCoOwner(resourceId, userIds); + // Close only on a clean apply; a partial failure keeps the modal open so + // the user can see what was rejected and retry. + if (await onApplyCoOwners(resourceId, { addUsers, removeUsers })) { + setOpen(false); + } } finally { - setPendingAdds([]); setApplying(false); } }; const handleCancel = () => { - setPendingAdds([]); + setSelectedOwners(ownersList); setOpen(false); }; const filterOption = (input, option) => (option?.label ?? "").toLowerCase().includes(input.toLowerCase()); - const combinedList = [ - ...ownersList, - ...pendingAdds.filter( - (pending) => - !ownersList.some( - (owner) => owner?.id?.toString() === pending?.id?.toString(), - ), - ), - ]; - return ( - {loading || applying ? ( + {loading ? ( ) : ( <> @@ -134,77 +136,56 @@ function CoOwnerManagement({ }))} /> Co-Owners - {combinedList.length > 0 ? ( + {selectedOwners.length > 0 ? ( { - const isPending = pendingAdds.some( - (u) => u?.id?.toString() === item?.id?.toString(), - ); - return ( - - } - onClick={() => handleRemovePending(item?.id)} - aria-label={`Remove pending co-owner ${item?.email}`} + dataSource={selectedOwners} + renderItem={(item) => ( + 1 && ( +
event.stopPropagation()} + role="none" + > + } + onConfirm={() => handleRemove(item?.id)} + > +
+ ) + } + > + + } /> - ) : ( - totalOwners > 1 && ( -
event.stopPropagation()} - role="none" - > - } - onConfirm={() => handleRemoveExisting(item?.id)} - > -
- ) - ) + + {item.email} + + } - > - - } - /> - - {item.email} - - - } - /> -
- ); - }} + /> +
+ )} /> ) : ( No co-owners yet @@ -218,13 +199,12 @@ function CoOwnerManagement({ CoOwnerManagement.propTypes = { open: PropTypes.bool.isRequired, setOpen: PropTypes.func.isRequired, - resourceId: PropTypes.string.isRequired, + resourceId: PropTypes.string, resourceType: PropTypes.string.isRequired, allUsers: PropTypes.array, coOwners: PropTypes.array, loading: PropTypes.bool, - onAddCoOwner: PropTypes.func.isRequired, - onRemoveCoOwner: PropTypes.func.isRequired, + onApplyCoOwners: PropTypes.func.isRequired, }; export { CoOwnerManagement }; diff --git a/frontend/src/components/widgets/co-owner-management/CoOwnerModal.jsx b/frontend/src/components/widgets/co-owner-management/CoOwnerModal.jsx index e66be15223..ccadfd0b82 100644 --- a/frontend/src/components/widgets/co-owner-management/CoOwnerModal.jsx +++ b/frontend/src/components/widgets/co-owner-management/CoOwnerModal.jsx @@ -21,10 +21,8 @@ function CoOwnerModal({ coOwner, resourceType }) { resourceType={resourceType} allUsers={coOwner.coOwnerAllUsers} coOwners={coOwner.coOwnerData.coOwners} - createdBy={coOwner.coOwnerData.createdBy} loading={coOwner.coOwnerLoading} - onAddCoOwner={coOwner.onAddCoOwner} - onRemoveCoOwner={coOwner.onRemoveCoOwner} + onApplyCoOwners={coOwner.onApplyCoOwners} /> ); } diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index 5a3b7e5ef1..c3fee93d10 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -2,14 +2,47 @@ import { useCallback, useRef, useState } from "react"; import { useExceptionHandler } from "./useExceptionHandler"; +/** + * Summarize one Apply into a single alert. + * + * Failures carry the user object rather than the id, so an owner who has since + * left the org — and is therefore missing from the org member list — is still + * named by email. + */ +function buildApplyAlert( + addUsers, + removeUsers, + failed, + lastError, + handleException, +) { + const total = addUsers.length + removeUsers.length; + if (failed.length === total) { + return handleException(lastError, "Unable to update co-owners"); + } + const failedIds = new Set(failed.map((user) => String(user?.id))); + const done = (users) => + users.filter((user) => !failedIds.has(String(user?.id))).length; + const parts = []; + if (done(addUsers)) { + parts.push(`${done(addUsers)} added`); + } + if (done(removeUsers)) { + parts.push(`${done(removeUsers)} removed`); + } + const summary = `Co-owners updated: ${parts.join(", ")}`; + if (failed.length === 0) { + return { type: "success", content: summary }; + } + const failedNames = failed.map((user) => user?.email || user?.id).join(", "); + return { type: "warning", content: `${summary}. Failed for: ${failedNames}` }; +} + function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { const handleException = useExceptionHandler(); const [coOwnerOpen, setCoOwnerOpen] = useState(false); - const [coOwnerData, setCoOwnerData] = useState({ - coOwners: [], - createdBy: null, - }); + const [coOwnerData, setCoOwnerData] = useState({ coOwners: [] }); const [coOwnerLoading, setCoOwnerLoading] = useState(false); const [coOwnerAllUsers, setCoOwnerAllUsers] = useState([]); const [coOwnerResourceId, setCoOwnerResourceId] = useState(null); @@ -25,10 +58,7 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { try { const res = await service.getSharedUsers(resourceId); if (latestRequestRef.current !== requestId) return; - setCoOwnerData({ - coOwners: res.data?.co_owners || [], - createdBy: res.data?.created_by || null, - }); + setCoOwnerData({ coOwners: res.data?.co_owners || [] }); } catch (err) { if (latestRequestRef.current !== requestId) return; if (err?.response?.status === 404) { @@ -74,7 +104,6 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { setCoOwnerAllUsers(userList); setCoOwnerData({ coOwners: sharedUsersResponse.data?.co_owners || [], - createdBy: sharedUsersResponse.data?.created_by || null, }); } catch (err) { if (latestRequestRef.current !== requestId) return; @@ -91,73 +120,40 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { [service, setAlertDetails, handleException], ); - const onAddCoOwner = useCallback( - async (resourceId, userIdOrIds) => { + const onApplyCoOwners = useCallback( + async (resourceId, { addUsers = [], removeUsers = [] }) => { const requestId = latestRequestRef.current; - const isBatch = Array.isArray(userIdOrIds); - const userIds = isBatch ? userIdOrIds : [userIdOrIds]; - // Attempt every id independently — a mid-batch failure must not drop the - // remaining ids or contradict the refreshed modal state. - const failedIds = []; + // Attempt every user independently — one rejection must not drop the rest + // or leave the modal contradicting the server. + const failed = []; let lastError = null; - for (const userId of userIds) { - try { - await service.addCoOwner(resourceId, userId); - } catch (err) { - failedIds.push(userId); - lastError = err; + const run = async (users, call) => { + for (const user of users) { + try { + await call(user.id); + } catch (err) { + failed.push(user); + lastError = err; + } } - } + }; + // Adds first: the backend rejects removing the last owner, so a one-shot + // owner swap has to grow the roster before it shrinks it. + await run(addUsers, (id) => service.addCoOwner(resourceId, id)); + await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); // Reconverge the modal on true server state regardless of partial outcome. await refreshCoOwnerData(resourceId, requestId); onListRefresh?.(); - - const succeeded = userIds.length - failedIds.length; - if (failedIds.length === 0) { - setAlertDetails({ - type: "success", - content: isBatch - ? "Co-owners added successfully" - : "Co-owner added successfully", - }); - } else if (succeeded === 0) { - setAlertDetails(handleException(lastError, "Unable to add co-owner")); - } else { - const failedEmails = coOwnerAllUsers - .filter((user) => failedIds.includes(user.id)) - .map((user) => user.email); - setAlertDetails({ - type: "warning", - content: `Added ${succeeded} of ${userIds.length} co-owners. Failed for: ${ - failedEmails.join(", ") || failedIds.join(", ") - }`, - }); - } - }, - [ - service, - refreshCoOwnerData, - onListRefresh, - setAlertDetails, - handleException, - coOwnerAllUsers, - ], - ); - - const onRemoveCoOwner = useCallback( - async (resourceId, userId) => { - const requestId = latestRequestRef.current; - try { - await service.removeCoOwner(resourceId, userId); - setAlertDetails({ - type: "success", - content: "Co-owner removed successfully", - }); - await refreshCoOwnerData(resourceId, requestId); - onListRefresh?.(); - } catch (err) { - setAlertDetails(handleException(err, "Unable to remove co-owner")); - } + setAlertDetails( + buildApplyAlert( + addUsers, + removeUsers, + failed, + lastError, + handleException, + ), + ); + return failed.length === 0; }, [ service, @@ -176,8 +172,7 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { coOwnerAllUsers, coOwnerResourceId, handleCoOwner, - onAddCoOwner, - onRemoveCoOwner, + onApplyCoOwners, }; } From c1d30955e58e3c9adb170e2e605c71ca5e432994 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 18:39:26 +0530 Subject: [PATCH 05/28] UN-3494 [FIX] Address review findings on sharing notifications MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Group revoke no longer mails members who kept access another way. The revoke recipient list now runs through the same effective-access filter the direct path uses, with owners folded in — compute_effective_members excludes them by design, and the sharer is usually a member of the group they shared with, so revoking told the owner their own access was removed and pointed them at the dashboard. - _get_user is org-scoped through OrganizationMember, the one unscoped query left on this tenant path. Service accounts are kept so a platform-account share still notifies. - _notify_shared_users is wrapped: the share has already committed by the time it runs, so a raising seam or a DB hiccup must not 500 a share that worked. - _users_left_without_access short-circuits on shared_to_org — nobody lost access, and answering it otherwise hydrates every member of the org. - _notification_context loses its duplicate copy and uses the OwnerManagementMixin definition every host already inherits. - Logs the Flipt decision, and how many recipients were dropped versus requested, so a missing email is diagnosable. - Docstrings corrected: the mixin is not axis-agnostic, transport resolves per resource id not per org, the flag is evaluated once at enqueue, and delivery is at-least-once. - Sonar S7632: the noqa directive carried trailing prose. - Co-owner apply no longer overwrites the resource-gone alert with its summary. Co-Authored-By: Claude Opus 5 --- backend/permissions/membership_views.py | 5 + backend/permissions/resource_share_views.py | 57 ++++---- .../group_notification_service.py | 127 +++++++++++++----- backend/tenant_account_v2/internal_views.py | 2 +- .../tenant_account_v2/share_notifications.py | 27 ++-- frontend/src/hooks/useCoOwnerManagement.jsx | 11 +- workers/notification/tasks.py | 7 +- 7 files changed, 159 insertions(+), 77 deletions(-) diff --git a/backend/permissions/membership_views.py b/backend/permissions/membership_views.py index e9ade97e9c..329c6a3aa5 100644 --- a/backend/permissions/membership_views.py +++ b/backend/permissions/membership_views.py @@ -86,6 +86,11 @@ def _owner_refs(resource: Any) -> list[dict[str, Any]]: # --- notifications: reuse the user-sharing service, best-effort --- def _notification_context(self, resource: Any) -> tuple[str, str] | None: + """``(resource_type, resource_name)``, or ``None`` if not notifiable. + + Also used by ``ResourceShareManagementMixin``, which every host mixes + in alongside this one. + """ if not notification_plugin or not self.notification_resource_name_field: return None resource_type = self.get_notification_resource_type(resource) diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 3b13cdc988..fbc61de1b7 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -1,9 +1,9 @@ """Shared share-management surface for resource ViewSets. -The mixin is **axis-agnostic** — it reads the sharing "axes" named in -``_SUPPORTED_SHARE_AXES``. ``shared_users`` is the direct-viewer axis, backed by -VIEWER membership rows, while ``shared_groups`` is stored polymorphically in -``ResourceGroupShare`` (not an M2M) and routed through the sharing helpers. +The mixin reads the sharing "axes" named in ``_SUPPORTED_SHARE_AXES``. +``shared_users`` is the direct-viewer axis, backed by VIEWER membership rows, +while ``shared_groups`` is stored polymorphically in ``ResourceGroupShare`` +(not an M2M) and routed through the sharing helpers. """ import logging @@ -59,25 +59,6 @@ def _coerce_id_list(axis: str, value: Any) -> list[int]: return coerced -def _notification_context(view: Any, instance: Any) -> tuple[str, str] | None: - """Resolve ``(resource_type, resource_name)`` for the email senders. - - ``None`` when the plugin is absent or the host ViewSet has not opted in by - setting ``notification_resource_name_field`` and overriding - ``get_notification_resource_type`` (both declared on - ``OwnerManagementMixin``, which every share host also mixes in). - """ - name_field = getattr(view, "notification_resource_name_field", None) - resolve_type = getattr(view, "get_notification_resource_type", None) - if not notification_plugin or not name_field or resolve_type is None: - return None - resource_type = resolve_type(instance) - resource_name = getattr(instance, name_field, None) - if resource_type is None or not resource_name: - return None - return resource_type, resource_name - - def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: """Narrow ``users`` to those with no remaining access to ``instance``. @@ -87,6 +68,11 @@ def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: """ if not users: return [] + if getattr(instance, "shared_to_org", False): + # Org-wide share still covers everyone — nobody lost access, and + # answering it via ``compute_effective_members`` would hydrate every + # member of the org to say so. + return [] from tenant_account_v2.sharing_helpers import compute_effective_members retained = {member["user_id"] for member in compute_effective_members(instance)} @@ -180,19 +166,24 @@ def _notify_shared_users( actor: Any, /, ) -> None: - """Email users granted or denied direct access. + """Email users granted or denied direct access. Best-effort. Resource type and name come from the host's ``OwnerManagementMixin`` - seam, so every share host is covered without an override. + seam. The share has already committed by the time this runs, so no + failure here — a raising seam, a dropped DB connection — may surface + as a 500 on a share that succeeded. """ - context = _notification_context(self, instance) - if context is None: - return - if added: - _send_share_notification(instance, context, added, actor) - revoked = _users_left_without_access(instance, removed) - if revoked: - _send_revoke_notification(instance, context, revoked, actor) + try: + context = self._notification_context(instance) # type: ignore[attr-defined] + if context is None: + return + if added: + _send_share_notification(instance, context, added, actor) + revoked = _users_left_without_access(instance, removed) + if revoked: + _send_revoke_notification(instance, context, revoked, actor) + except Exception: + logger.exception("Failed to send share notifications for %s", instance.pk) @action(detail=True, methods=["get"], url_path="effective-members") def effective_members(self, request: Request, pk: str | None = None) -> Response: diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index ccdc3d9989..7030d642f2 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -12,6 +12,7 @@ from __future__ import annotations import logging +from dataclasses import dataclass from typing import TYPE_CHECKING, Any from account_v2.models import Organization, User @@ -55,6 +56,15 @@ class ResourceNotFoundError(Exception): """The shared resource no longer exists, or is not in the given org.""" +@dataclass(frozen=True) +class _SharedResource: + """A resolved resource, reused across every group email in one task.""" + + instance: Any + name: str + type: str | None + + def send_resource_shared( *, organization: Organization, @@ -72,43 +82,30 @@ def send_resource_shared( service = _service() if service is None: return - actor = _get_user(actor_id) - resource, resource_name, resource_type = _load_resource( - organization, resource_kind, resource_id - ) - if actor is None or resource_type is None: + actor = _get_user(organization, actor_id) + shared = _load_resource(organization, resource_kind, resource_id) + if actor is None or shared.type is None: logger.info( "group-notification: skipping resource share for %s/%s " "(actor_found=%s resource_type=%s)", resource_kind, resource_id, actor is not None, - resource_type, + shared.type, ) return + retained = _retained_user_ids(shared.instance, share_action) for group in _groups_in_org(organization, group_ids): - recipients = _live_member_users( - organization, group.memberships.values_list("user_id", flat=True) - ) + recipients = _group_recipients(organization, group, retained) logger.info( - "group-notification: task=%s group_id=%s action=%s recipient_count=%d", - "notify_resource_shared_with_group", + "group-notification: task=notify_resource_shared_with_group " + "group_id=%s action=%s recipient_count=%d", group.pk, share_action, len(recipients), ) - if not recipients: - continue - service.send_group_resource_shared_notification( - resource_type=resource_type, - resource_name=resource_name, - resource_id=str(resource.pk), - group_name=group.name, - shared_by=actor, - shared_to=recipients, - resource_instance=resource, - share_action=ShareAction(share_action).value, - ) + if recipients: + _mail_group(service, group, recipients, shared, actor, share_action) def send_membership_changed( @@ -129,7 +126,7 @@ def send_membership_changed( service = _service() if service is None: return - actor = _get_user(actor_id) + actor = _get_user(organization, actor_id) group = _groups_in_org(organization, [group_id]).first() if actor is None or group is None: logger.info( @@ -167,8 +164,67 @@ def _service() -> Any | None: return notification_plugin["service_class"]() -def _get_user(user_id: int) -> User | None: - return User.objects.filter(pk=user_id).first() +def _get_user(organization: Organization, user_id: int) -> User | None: + """The actor, re-validated against the org like every recipient is. + + Service accounts are kept: a share performed by a platform account must + still notify the group. + """ + member = ( + OrganizationMember.objects.filter(organization=organization, user_id=user_id) + .select_related("user") + .first() + ) + return member.user if member else None + + +def _retained_user_ids(resource: Any, share_action: str) -> set[int]: + """Users who still reach ``resource``; empty on the share direction. + + A revoked group's members may keep access through another group, a direct + share or an org-wide share — telling them it was removed would be wrong, + and the revoke email also repoints their CTA at the dashboard. Owners sit + outside ``compute_effective_members`` by design, so add them back: an owner + in the revoked group has lost nothing. + """ + if share_action != ShareAction.REVOKED.value: + return set() + from tenant_account_v2.sharing_helpers import compute_effective_members + + return {member["user_id"] for member in compute_effective_members(resource)} | { + owner.pk for owner in resource.owners() + } + + +def _group_recipients( + organization: Organization, group: OrganizationGroup, retained: set[int] +) -> list[User]: + """Live members of ``group`` who did not keep access via ``retained``.""" + users = _live_member_users( + organization, group.memberships.values_list("user_id", flat=True) + ) + return [user for user in users if user.pk not in retained] + + +def _mail_group( + service: Any, + group: OrganizationGroup, + recipients: list[User], + shared: _SharedResource, + actor: User, + share_action: str, +) -> None: + """Send one group's copy of the resource-share email.""" + service.send_group_resource_shared_notification( + resource_type=shared.type, + resource_name=shared.name, + resource_id=str(shared.instance.pk), + group_name=group.name, + shared_by=actor, + shared_to=recipients, + resource_instance=shared.instance, + share_action=ShareAction(share_action).value, + ) def _groups_in_org( @@ -185,20 +241,29 @@ def _live_member_users(organization: Organization, user_ids: Iterable[int]) -> l Service accounts are excluded, matching ``compute_effective_members``. """ + requested = list(user_ids) memberships = OrganizationMember.objects.filter( - organization=organization, user_id__in=list(user_ids) + organization=organization, user_id__in=requested ).select_related("user") - return [ + users = [ m.user for m in memberships if not getattr(m.user, "is_service_account", False) and m.user.email ] + if len(users) != len(requested): + logger.info( + "group-notification: dropped %d of %d recipients " + "(left the org / service account / no email)", + len(requested) - len(users), + len(requested), + ) + return users def _load_resource( organization: Organization, kind: str, resource_id: str -) -> tuple[Any, str, str | None]: - """Resolve the shared resource to ``(instance, display name, plugin type)``. +) -> _SharedResource: + """Resolve the shared resource for the email senders. Raises: ResourceNotFoundError: the descriptor, model, or row is missing — the @@ -220,7 +285,7 @@ def _load_resource( if resource is None: raise ResourceNotFoundError(f"{kind} {resource_id} not found in organization") name = getattr(resource, descriptor.name_field, "") or "" - return resource, name, _resource_type_for(descriptor, resource) + return _SharedResource(resource, name, _resource_type_for(descriptor, resource)) def _resource_type_for(descriptor: ShareableResource, resource: Any) -> str | None: diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index 0e959c9b5d..c5797b53c0 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -6,7 +6,7 @@ email plugin — needs it. Failure contract: **any** unhandled problem must surface as non-2xx so the -queue redelivers. The one deliberate exception is a resource that no longer +worker retries. The one deliberate exception is a resource that no longer exists, which returns 200 — retrying that can only fail again. """ diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 526f926a04..50155b0aba 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -6,13 +6,16 @@ The sending itself runs in ``workers/``, which is Django-free, so the worker task is a thin HTTP shim back to :mod:`tenant_account_v2.internal_views`; the -backend does the ORM and plugin work. Transport is resolved per-org by the same -``resolve_transport`` gate the execution path uses — the PG queue where that is -enabled, Celery otherwise. - -The whole feature sits behind its own Flipt flag and fails closed everywhere: a -blind Flipt, a missing org, or any dispatch error means no notification, never -a broken share. +backend does the ORM and plugin work. Transport is resolved per resource/group +id by the same ``resolve_transport`` gate the execution path uses — the PG +queue where that is enabled, Celery otherwise. + +The group feature sits behind its own Flipt flag, evaluated once here at +enqueue: a blind Flipt, a missing org, or any dispatch error means no +notification, never a broken share. Turning the flag off stops new enqueues; a +message already queued still delivers. Direct-user share and revoke mail +(``ResourceShareManagementMixin``) is not on this flag — like the co-owner mail +it reuses, it is gated only by the cloud ``ENABLE_EMAIL_NOTIFICATIONS`` setting. """ from __future__ import annotations @@ -152,9 +155,14 @@ def _feature_enabled(organization_id: str) -> bool: # Parse exactly as FliptClient does (``.lower()``, no ``.strip()``) so the # two can never disagree on a value like " true". if os.environ.get("FLIPT_SERVICE_AVAILABLE", "false").lower() != "true": + logger.warning( + "group-notification: FLIPT_SERVICE_AVAILABLE != true (Flipt blind) " + "for org %s; skipping", + organization_id, + ) return False try: - return bool( + enabled = bool( check_feature_flag_status( flag_key=GROUP_NOTIFICATION_FLAG_KEY, entity_id=organization_id, @@ -168,6 +176,9 @@ def _feature_enabled(organization_id: str) -> bool: exc_info=True, ) return False + if not enabled: + logger.info("group-notification: flag off for org %s; skipping", organization_id) + return enabled def _organization_slug(obj: Any) -> str | None: diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index c3fee93d10..ed00733da7 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -54,6 +54,8 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // branch) after the user has moved to a different resource. Mutation // callers pass the token captured BEFORE their POSTs so a modal switch // during the mutation itself is caught too, not just one mid-refresh. + // Returns true when the resource turned out to be gone, so the caller can + // leave that alert standing instead of overwriting it with its own. async (resourceId, requestId = latestRequestRef.current) => { try { const res = await service.getSharedUsers(resourceId); @@ -69,7 +71,7 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { content: "This resource is no longer accessible. It may have been removed or your access has been revoked.", }); - return; + return true; } setAlertDetails( handleException(err, "Unable to refresh co-owner data"), @@ -142,7 +144,12 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { await run(addUsers, (id) => service.addCoOwner(resourceId, id)); await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); // Reconverge the modal on true server state regardless of partial outcome. - await refreshCoOwnerData(resourceId, requestId); + const gone = await refreshCoOwnerData(resourceId, requestId); + if (gone) { + // The refresh already closed the modal, refreshed the list and raised + // its own alert — an apply summary on top of it would only mislead. + return true; + } onListRefresh?.(); setAlertDetails( buildApplyAlert( diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 2de8d237fd..1f8b60a3b4 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -489,7 +489,10 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) silently unsent email. On the PG transport the raise also leaves the message on the queue for redelivery, bounded by the consumer's attempt cap. - A 4xx is not retried — a rejected payload will be rejected again. + A sub-500 response ends the in-process attempts — a rejected payload will + be rejected again. Delivery is at-least-once: a response lost after the + backend already sent re-posts the same payload, and the send path writes + nothing, so the only effect is a duplicate email. """ base_url = os.getenv("INTERNAL_API_BASE_URL") api_key = os.getenv("INTERNAL_SERVICE_API_KEY") @@ -510,7 +513,7 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) try: with httpx.Client(transport=httpx.HTTPTransport(retries=2)) as client: response = client.post(url, headers=headers, json=payload, timeout=30.0) - except Exception as e: # noqa: BLE001 - transport failure, retry below + except Exception as e: # noqa: BLE001 last_error = f"exception={e!r}" else: if response.status_code == 200: From 9e9f57c4addc13bc63706b196338b6a8bd77e40b Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 18:39:33 +0530 Subject: [PATCH 06/28] UN-3494 [MISC] Drop the local-only worker-pg-notification compose service The PG-queue notification consumer is local dev config and does not belong in the PR. The k8s chart already carries workerPgNotification from UN-3445 (#1688), which is the real deployment surface. Co-Authored-By: Claude Opus 5 --- docker/docker-compose.yaml | 36 ------------------------------------ 1 file changed, 36 deletions(-) diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index d74dc2bf43..9a8db90afe 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -827,42 +827,6 @@ services: profiles: - pg-queue - # Notification consumer — webhook POSTs and the group-share emails (UN-3494). - # Without this, a notification enqueued on PG is durably stored and never run. - # Every task here is one short outbound HTTP call, so it stays light. - worker-pg-notification: - image: unstract/worker-unified:${VERSION} - container_name: unstract-worker-pg-notification - restart: unless-stopped - command: ["pg-queue-consumer"] - ports: - - "8101:8090" - env_file: - - ../workers/.env - - ./essentials.env - depends_on: - - db - - redis - environment: - - ENVIRONMENT=development - - APPLICATION_NAME=unstract-worker-pg-notification - - WORKER_BARRIER_BACKEND=pg - - WORKER_PG_QUEUE_CONSUMER_WORKER_TYPE=notification - - WORKER_PG_QUEUE_CONSUMER_QUEUE=notifications - - WORKER_PG_QUEUE_CONSUMER_HEALTH_PORT=8090 - - WORKER_PG_QUEUE_CONSUMER_CONCURRENCY=${PG_NOTIFICATION_CONCURRENCY:-4} - # One internal-API call (30s timeout) plus a SendGrid batch for the - # largest group, with headroom. Health-stale sits at or above it. - - WORKER_PG_QUEUE_CONSUMER_VT_SECONDS=${PG_NOTIFICATION_VT_SECONDS:-120} - - WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS=${PG_NOTIFICATION_HEALTH_STALE_SECONDS:-180} - labels: - - traefik.enable=false - volumes: - - ./workflow_data:/data - - ${TOOL_REGISTRY_CONFIG_SRC_PATH}:/data/tool_registry_config - profiles: - - pg-queue - # Reaper / orchestrator — leader-elected loop. Run exactly ONE instance (it # elects a single leader via pg_orchestrator_lock; extra replicas idle as # standby). Besides barrier-orphan recovery it runs the PG scheduler tick From c663694c02c001756b452d57e4d357c3022cbdad Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 18:55:35 +0530 Subject: [PATCH 07/28] UN-3494 [FIX] Exclude members who joined a group after its access was revoked A revoke resolves recipients from the group's live membership at delivery time, so anyone who joined between the click and the send was told their access was removed for a group through which they never held it. Normally a few seconds; on the PG transport with no consumer deployed the backlog can sit far longer. The revoke now carries the timestamp of the change and delivery drops memberships created after it. One string on the payload rather than the frozen member list, which would grow with the group. Co-Authored-By: Claude Opus 5 --- .../group_notification_service.py | 24 +++++++++++++---- backend/tenant_account_v2/internal_views.py | 4 ++- .../tenant_account_v2/share_notifications.py | 27 +++++++++++++------ workers/notification/tasks.py | 5 +++- 4 files changed, 45 insertions(+), 15 deletions(-) diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 7030d642f2..e16139617b 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -26,6 +26,7 @@ if TYPE_CHECKING: from collections.abc import Iterable + from datetime import datetime logger = logging.getLogger(__name__) @@ -73,11 +74,13 @@ def send_resource_shared( resource_kind: str, resource_id: str, share_action: str = ShareAction.SHARED.value, + revoked_at: datetime | None = None, ) -> None: """Mail every current member of each group whose resource access changed. One email per group, so ``group_name`` in the template is always the group - the recipient actually belongs to. ``share_action`` picks the wording. + the recipient actually belongs to. ``share_action`` picks the wording, and + on a revoke ``revoked_at`` bounds who counts as "current". """ service = _service() if service is None: @@ -96,7 +99,7 @@ def send_resource_shared( return retained = _retained_user_ids(shared.instance, share_action) for group in _groups_in_org(organization, group_ids): - recipients = _group_recipients(organization, group, retained) + recipients = _group_recipients(organization, group, retained, revoked_at) logger.info( "group-notification: task=notify_resource_shared_with_group " "group_id=%s action=%s recipient_count=%d", @@ -197,11 +200,22 @@ def _retained_user_ids(resource: Any, share_action: str) -> set[int]: def _group_recipients( - organization: Organization, group: OrganizationGroup, retained: set[int] + organization: Organization, + group: OrganizationGroup, + retained: set[int], + joined_before: datetime | None = None, ) -> list[User]: - """Live members of ``group`` who did not keep access via ``retained``.""" + """Live members of ``group`` who did not keep access via ``retained``. + + ``joined_before`` (a revoke's timestamp) drops anyone who joined after the + access was taken away: they never held it through this group, so a + revocation notice would be about access they never had. + """ + memberships = group.memberships + if joined_before is not None: + memberships = memberships.filter(created_at__lte=joined_before) users = _live_member_users( - organization, group.memberships.values_list("user_id", flat=True) + organization, memberships.values_list("user_id", flat=True) ) return [user for user in users if user.pk not in retained] diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index c5797b53c0..47c292580f 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -37,10 +37,12 @@ class ResourceSharedWithGroupSerializer(serializers.Serializer): actor_id = serializers.IntegerField() resource_kind = serializers.CharField() resource_id = serializers.CharField() - # Defaulted so messages enqueued before this field existed still validate. share_action = serializers.ChoiceField( choices=[a.value for a in ShareAction], default=ShareAction.SHARED.value ) + # Revoke only: members who joined after this are excluded from the mail. + # Nullable because the worker sends the key on both directions. + revoked_at = serializers.DateTimeField(allow_null=True, default=None) class GroupMembershipChangedSerializer(serializers.Serializer): diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 50155b0aba..31d94dfec2 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -26,6 +26,8 @@ from enum import StrEnum from typing import TYPE_CHECKING, Any +from django.utils import timezone + from tenant_account_v2.shareable_resources import kind_for_instance from unstract.core.data_models import is_pg_transport from unstract.flags.feature_flag import check_feature_flag_status @@ -94,6 +96,12 @@ def _notify_group_share( lookup, so offboarding safety costs nothing. Unlike a membership removal, revoking a group's access leaves the group and its members intact, so the fresh lookup still finds everyone who needs telling. + + A revoke carries ``revoked_at`` so that fresh lookup can still exclude + anyone who joined the group *after* the access was taken away — they never + held it through this group, and the queue can lag (see the PG rollout + ordering note). One timestamp rather than the whole member list, which + would grow the payload with the group. """ group_ids = sorted(group.pk for group in groups) if not group_ids: @@ -102,16 +110,19 @@ def _notify_group_share( kind = kind_for_instance(resource) if not organization_id or kind is None or not _feature_enabled(organization_id): return + kwargs: dict[str, Any] = { + "group_ids": group_ids, + "actor_id": actor.pk, + "resource_kind": kind, + "resource_id": str(resource.pk), + "share_action": str(share_action), + "organization_id": organization_id, + } + if share_action is ShareAction.REVOKED: + kwargs["revoked_at"] = timezone.now().isoformat() _dispatch_quietly( task_name=NOTIFY_RESOURCE_SHARED_TASK, - kwargs={ - "group_ids": group_ids, - "actor_id": actor.pk, - "resource_kind": kind, - "resource_id": str(resource.pk), - "share_action": str(share_action), - "organization_id": organization_id, - }, + kwargs=kwargs, organization_id=organization_id, entity_id=str(resource.pk), ) diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 1f8b60a3b4..a85ed0c2aa 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -541,10 +541,12 @@ def notify_resource_shared_with_group( resource_id: str, organization_id: str, share_action: str = "shared", + revoked_at: str | None = None, ) -> None: """Email every current member of the groups whose access just changed. - ``share_action`` defaults so messages enqueued before it existed still run. + ``revoked_at`` is set on a revoke only; the backend uses it to skip members + who joined the group after the access was taken away. """ _post_group_notification( "resource-shared", @@ -555,6 +557,7 @@ def notify_resource_shared_with_group( "resource_kind": resource_kind, "resource_id": resource_id, "share_action": share_action, + "revoked_at": revoked_at, }, ) From 6ab867b0cde7e1a3dd74e4483637431a5420e6f2 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 19:00:55 +0530 Subject: [PATCH 08/28] UN-3494 [FIX] Skip a queued grant email when the group's access is already gone MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A grant enqueued before a revoke could still be delivered after it, mailing the resource name and id to members who can no longer reach the resource. Delivery now revalidates the live ResourceGroupShare on the grant direction and drops groups that no longer hold it. The revoke direction needs no equivalent check — its share row is gone by delivery, and _retained_user_ids already covers members who kept access another way. Co-Authored-By: Claude Opus 5 --- .../group_notification_service.py | 25 ++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index e16139617b..e577e9359f 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -98,7 +98,7 @@ def send_resource_shared( ) return retained = _retained_user_ids(shared.instance, share_action) - for group in _groups_in_org(organization, group_ids): + for group in _groups_to_mail(organization, group_ids, shared.instance, share_action): recipients = _group_recipients(organization, group, retained, revoked_at) logger.info( "group-notification: task=notify_resource_shared_with_group " @@ -199,6 +199,29 @@ def _retained_user_ids(resource: Any, share_action: str) -> set[int]: } +def _groups_to_mail( + organization: Organization, + group_ids: Iterable[int], + resource: Any, + share_action: str, +) -> Iterable[OrganizationGroup]: + """Groups from the payload that should still be mailed. + + On a grant, drop any group whose access was revoked between enqueue and + delivery: the mail carries the resource name and id, so announcing access + the group no longer holds discloses both to members who cannot reach it. + The revoke direction needs no such check — its share row is already gone, + and ``_retained_user_ids`` covers who kept access another way. + """ + groups = _groups_in_org(organization, group_ids) + if share_action != ShareAction.SHARED.value: + return groups + from tenant_account_v2.sharing_helpers import get_resource_share_groups + + live = {group.pk for group in get_resource_share_groups(resource)} + return [group for group in groups if group.pk in live] + + def _group_recipients( organization: Organization, group: OrganizationGroup, From 70b42c8126c7e9cf435e6974427ba412a3217080 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 19:44:33 +0530 Subject: [PATCH 09/28] UN-3494 [FIX] Stamp the revoke cutoff before the feature-flag round-trip revoked_at was captured after _feature_enabled(), so the window between the share-removal commit and the timestamp spanned a Flipt network call. A user joining the group inside it passed the cutoff and was mailed a revocation. Co-Authored-By: Claude Opus 5 --- backend/tenant_account_v2/share_notifications.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 31d94dfec2..f418be24e1 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -106,6 +106,11 @@ def _notify_group_share( group_ids = sorted(group.pk for group in groups) if not group_ids: return + # Stamped before the Flipt round-trip below: a member joining inside that + # window would be mailed a revocation for access they never held. + revoked_at = ( + timezone.now().isoformat() if share_action is ShareAction.REVOKED else None + ) organization_id = _organization_slug(resource) kind = kind_for_instance(resource) if not organization_id or kind is None or not _feature_enabled(organization_id): @@ -118,8 +123,8 @@ def _notify_group_share( "share_action": str(share_action), "organization_id": organization_id, } - if share_action is ShareAction.REVOKED: - kwargs["revoked_at"] = timezone.now().isoformat() + if revoked_at is not None: + kwargs["revoked_at"] = revoked_at _dispatch_quietly( task_name=NOTIFY_RESOURCE_SHARED_TASK, kwargs=kwargs, From 3f7016e8da967fff9596ab153a1922379cfb67d6 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 5 Aug 2026 20:26:29 +0530 Subject: [PATCH 10/28] UN-3494 [TEST] Cover share/revoke notifications for users and groups Enqueue side (unit tier, no DB): payload shape, the revoked_at stamp landing before the Flipt round-trip, and the skip/swallow paths. Delivery side (integration tier): recipient selection - the live re-read on a grant, the revoked_at cutoff, org scoping and retained access - plus the direct-user share/revoke wiring on the share endpoint. Co-Authored-By: Claude Opus 5 --- .../tests/test_share_notifications.py | 104 ++++++++++ .../test_share_notification_dispatch.py | 183 ++++++++++++++++++ backend/tenant_account_v2/tests.py | 123 +++++++++++- 3 files changed, 409 insertions(+), 1 deletion(-) create mode 100644 backend/permissions/tests/test_share_notifications.py create mode 100644 backend/tenant_account_v2/test_share_notification_dispatch.py diff --git a/backend/permissions/tests/test_share_notifications.py b/backend/permissions/tests/test_share_notifications.py new file mode 100644 index 0000000000..3c7b7f1794 --- /dev/null +++ b/backend/permissions/tests/test_share_notifications.py @@ -0,0 +1,104 @@ +"""Integration tests for direct-user share/revoke email wiring (UN-3494). + +The ``share/`` endpoint's ``shared_users`` axis mails users who gained or lost +direct access. Both notification seams are mocked, so these pin the wiring and +the payload — who is mailed, with what, and that a failing send never breaks a +share that already committed — not template or transport behavior. The group +axis is covered by ``ResourceShareNotificationTests`` in +``tenant_account_v2.tests``. + +DB-backed (Django ``TestCase``), so ``backend/conftest.py`` auto-marks these +``integration`` and the rig runs them in ``integration-backend``. +""" + +from unittest.mock import Mock, patch + +from account_v2.models import User +from django.test import TestCase +from rest_framework import status +from rest_framework.response import Response +from rest_framework.test import APIRequestFactory, force_authenticate +from workflow_manager.workflow_v2.models.workflow import Workflow +from workflow_manager.workflow_v2.views import WorkflowViewSet + +from permissions.roles import ResourceRole +from permissions.tests.base import CoOwnerOrgTestMixin + + +class DirectShareNotificationWiringTests(CoOwnerOrgTestMixin, TestCase): + """``POST share/`` mails users whose direct access was granted or revoked.""" + + def setUp(self) -> None: + self._seed_org() + self.workflow = Workflow.objects.create( + workflow_name="wf-1", organization=self.org, created_by=self.owner + ) + self.workflow.memberships.create(user=self.owner, role=ResourceRole.OWNER) + self.factory = APIRequestFactory() + self.service = Mock() + plugin = {"service_class": Mock(return_value=self.service)} + for p in ( + # The sender lives in the share mixin; ``_notification_context`` + # gates on the membership_views copy, so both need the plugin. + patch("permissions.resource_share_views.notification_plugin", plugin), + patch("permissions.membership_views.notification_plugin", plugin), + patch.object( + WorkflowViewSet, + "get_notification_resource_type", + return_value="workflow", + ), + ): + p.start() + self.addCleanup(p.stop) + + def _share(self, actor: User, payload: dict) -> Response: + view = WorkflowViewSet.as_view({"post": "share"}) + request = self.factory.post("/x/", payload, format="json") + force_authenticate(request, user=actor) + return view(request, pk=str(self.workflow.pk)) + + def test_granting_direct_access_fires_sharing_notification(self) -> None: + response = self._share(self.owner, {"shared_users": [self.viewer.pk]}) + self.assertEqual(response.status_code, status.HTTP_200_OK) + self.service.send_sharing_notification.assert_called_once() + kwargs = self.service.send_sharing_notification.call_args.kwargs + self.assertEqual(kwargs["resource_type"], "workflow") + self.assertEqual(kwargs["resource_name"], "wf-1") + self.assertEqual(kwargs["resource_id"], str(self.workflow.pk)) + self.assertEqual(kwargs["shared_by"], self.owner) + self.assertEqual([u.pk for u in kwargs["shared_to"]], [self.viewer.pk]) + self.assertEqual(kwargs["resource_instance"], self.workflow) + self.service.send_access_removed_notification.assert_not_called() + + def test_revoking_direct_access_fires_access_removed_notification(self) -> None: + self.workflow.memberships.create(user=self.viewer, role=ResourceRole.VIEWER) + response = self._share(self.owner, {"shared_users": []}) + self.assertEqual(response.status_code, status.HTTP_200_OK) + self.service.send_access_removed_notification.assert_called_once() + kwargs = self.service.send_access_removed_notification.call_args.kwargs + self.assertEqual(kwargs["resource_type"], "workflow") + self.assertEqual([u.pk for u in kwargs["removed_from"]], [self.viewer.pk]) + self.assertEqual(kwargs["removed_by"], self.owner) + self.assertEqual(kwargs["resource_id"], str(self.workflow.pk)) + self.service.send_sharing_notification.assert_not_called() + + def test_revoke_is_silent_when_the_user_keeps_access_another_way(self) -> None: + # Dropped from ``shared_users`` but still covered by the org-wide share — + # nothing was lost, so telling them it was removed would be wrong. + self.workflow.memberships.create(user=self.viewer, role=ResourceRole.VIEWER) + response = self._share(self.owner, {"shared_users": [], "shared_to_org": True}) + self.assertEqual(response.status_code, status.HTTP_200_OK) + self.service.send_access_removed_notification.assert_not_called() + + def test_notification_failure_does_not_break_the_share(self) -> None: + # The share commits before the mail goes out; a raising sender must not + # surface as a 500 on a share that succeeded. + self.service.send_sharing_notification.side_effect = RuntimeError("boom") + response = self._share(self.owner, {"shared_users": [self.viewer.pk]}) + self.assertEqual(response.status_code, status.HTTP_200_OK) + viewer_ids = set( + self.workflow.memberships.filter(role=ResourceRole.VIEWER).values_list( + "user_id", flat=True + ) + ) + self.assertIn(self.viewer.pk, viewer_ids) diff --git a/backend/tenant_account_v2/test_share_notification_dispatch.py b/backend/tenant_account_v2/test_share_notification_dispatch.py new file mode 100644 index 0000000000..8958b00e5d --- /dev/null +++ b/backend/tenant_account_v2/test_share_notification_dispatch.py @@ -0,0 +1,183 @@ +"""Unit tests for the group-notification enqueue side (UN-3494 / mfbt UNS-848). + +``share_notifications`` runs inside the user's share request: it builds the task +payload and hands it to the transport. Nothing here touches the ORM or sends +mail, so the module is patched at its three seams — ``kind_for_instance``, +``_feature_enabled`` and ``_dispatch`` — and these run in the rig's unit tier +with no Postgres. The transport itself is covered by ``pg_queue.tests`` and +``workflow_manager.workflow_v2.tests.test_transport``; the delivery side by +``ResourceShareNotificationTests`` in ``tenant_account_v2.tests``. +""" + +from __future__ import annotations + +from contextlib import contextmanager +from datetime import UTC, datetime, timedelta +from types import SimpleNamespace +from unittest.mock import patch + +import tenant_account_v2.share_notifications as sn + +_ACTOR = SimpleNamespace(pk=7) +_RESOURCE = SimpleNamespace( + pk="wf-1", organization=SimpleNamespace(organization_id="org-a") +) + + +def _group(pk: int) -> SimpleNamespace: + return SimpleNamespace(pk=pk) + + +@contextmanager +def _seams(*, enabled: bool = True, dispatch_raises: Exception | None = None): + """Patch the module's three outbound seams; yield the flag + dispatch mocks.""" + with ( + patch.object(sn, "kind_for_instance", return_value="workflow"), + patch.object(sn, "_feature_enabled", return_value=enabled) as flag, + patch.object(sn, "_dispatch", side_effect=dispatch_raises) as dispatch, + ): + yield flag, dispatch + + +class TestNotifyResourceGroupShareChanged: + def test_grant_dispatches_shared_payload_without_timestamp(self): + with _seams() as (_, dispatch): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(5), _group(2)], removed=[], actor=_ACTOR + ) + dispatch.assert_called_once() + call = dispatch.call_args.kwargs + assert call["task_name"] == sn.NOTIFY_RESOURCE_SHARED_TASK + assert call["organization_id"] == "org-a" + assert call["entity_id"] == "wf-1" + assert call["kwargs"] == { + "group_ids": [2, 5], # sorted, so the payload is stable + "actor_id": 7, + "resource_kind": "workflow", + "resource_id": "wf-1", + "share_action": "shared", + "organization_id": "org-a", + } + # A grant carries no cutoff — the delivery side mails every live member. + assert "revoked_at" not in call["kwargs"] + + def test_revoke_dispatches_revoked_payload_with_timestamp(self): + with _seams() as (_, dispatch): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[], removed=[_group(3)], actor=_ACTOR + ) + payload = dispatch.call_args.kwargs["kwargs"] + assert payload["share_action"] == "revoked" + assert payload["group_ids"] == [3] + # ISO-8601 string, not a datetime — the payload is JSON-serialized. + datetime.fromisoformat(payload["revoked_at"]) + + def test_revoked_at_is_stamped_before_the_flipt_round_trip(self): + # Regression (PR #2224): the stamp used to sit below ``_feature_enabled``, + # whose Flipt call is a network round-trip. Someone joining the group + # inside that window is mailed a revocation for access never held. + clock = [datetime(2026, 1, 1, 12, 0, tzinfo=UTC)] + + def _flipt(_org: str) -> bool: + clock[0] += timedelta(seconds=5) # stand-in for the Flipt round-trip + return True + + with ( + patch.object(sn.timezone, "now", side_effect=lambda: clock[0]), + patch.object(sn, "kind_for_instance", return_value="workflow"), + patch.object(sn, "_feature_enabled", side_effect=_flipt), + patch.object(sn, "_dispatch") as dispatch, + ): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[], removed=[_group(3)], actor=_ACTOR + ) + revoked_at = dispatch.call_args.kwargs["kwargs"]["revoked_at"] + assert revoked_at == datetime(2026, 1, 1, 12, 0, tzinfo=UTC).isoformat() + + def test_grant_and_revoke_dispatch_independently(self): + with _seams() as (_, dispatch): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(1)], removed=[_group(2)], actor=_ACTOR + ) + assert dispatch.call_count == 2 + actions = [c.kwargs["kwargs"]["share_action"] for c in dispatch.call_args_list] + assert actions == ["shared", "revoked"] + + def test_no_groups_skips_before_the_flag_check(self): + with _seams() as (flag, dispatch): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[], removed=[], actor=_ACTOR + ) + dispatch.assert_not_called() + flag.assert_not_called() # no Flipt call for a no-op share + + def test_flag_off_skips_dispatch(self): + with _seams(enabled=False) as (_, dispatch): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(1)], removed=[], actor=_ACTOR + ) + dispatch.assert_not_called() + + def test_unknown_resource_kind_skips_dispatch(self): + with ( + patch.object(sn, "kind_for_instance", return_value=None), + patch.object(sn, "_feature_enabled", return_value=True), + patch.object(sn, "_dispatch") as dispatch, + ): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(1)], removed=[], actor=_ACTOR + ) + dispatch.assert_not_called() + + def test_missing_organization_skips_dispatch(self): + orphan = SimpleNamespace(pk="wf-1", organization=None) + with _seams() as (_, dispatch): + sn.notify_resource_group_share_changed( + resource=orphan, added=[_group(1)], removed=[], actor=_ACTOR + ) + dispatch.assert_not_called() + + def test_dispatch_failure_never_reaches_the_caller(self): + # The share has already committed — losing its email must not 500 it. + with _seams(dispatch_raises=RuntimeError("queue down")): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(1)], removed=[], actor=_ACTOR + ) + + +class TestNotifyGroupMembershipChanged: + def test_membership_change_dispatches_user_ids_in_payload(self): + with _seams() as (_, dispatch): + sn.notify_group_membership_changed( + group=SimpleNamespace( + pk=9, organization=SimpleNamespace(organization_id="org-a") + ), + action=sn.MembershipAction.ADDED, + user_ids=[4, 1], + actor=_ACTOR, + ) + call = dispatch.call_args.kwargs + assert call["task_name"] == sn.NOTIFY_MEMBERSHIP_CHANGED_TASK + assert call["entity_id"] == "9" + assert call["kwargs"] == { + "group_id": 9, + "actor_id": 7, + "membership_action": "added", + # Unlike a share, the ids ride in the payload: on removal the rows + # are gone by delivery time. + "user_ids": [1, 4], + "organization_id": "org-a", + } + + def test_no_users_skips_before_the_flag_check(self): + with _seams() as (flag, dispatch): + sn.notify_group_membership_changed( + group=SimpleNamespace( + pk=9, organization=SimpleNamespace(organization_id="org-a") + ), + action=sn.MembershipAction.REMOVED, + user_ids=[], + actor=_ACTOR, + ) + dispatch.assert_not_called() + flag.assert_not_called() diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index a105593599..eef404389c 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -11,18 +11,24 @@ """ import secrets -from unittest.mock import patch +from datetime import timedelta +from unittest.mock import Mock, patch from account_v2.models import Organization, User from django.contrib.contenttypes.models import ContentType from django.core.exceptions import FieldDoesNotExist from django.test import TestCase +from django.utils import timezone from permissions.roles import ResourceRole from rest_framework.exceptions import PermissionDenied from rest_framework.test import APIRequestFactory, force_authenticate from utils.user_context import UserContext from workflow_manager.workflow_v2.models.workflow import Workflow +from tenant_account_v2.group_notification_service import ( + send_membership_changed, + send_resource_shared, +) from tenant_account_v2.group_views import OrganizationGroupViewSet from tenant_account_v2.models import ( GroupMembership, @@ -30,6 +36,7 @@ OrganizationMember, ResourceGroupShare, ) +from tenant_account_v2.share_notifications import MembershipAction, ShareAction from tenant_account_v2.shareable_resources import SHAREABLE_RESOURCES from tenant_account_v2.sharing_helpers import ( ShareAuthorizationService, @@ -482,3 +489,117 @@ def test_descriptors_resolve_and_fields_exist(self) -> None: f"{resource.kind}.{attr}={field_name!r} is not a field on " f"{resource.app_label}.{resource.model_name}" ) + + +class ResourceShareNotificationTests(GroupSharingTestBase): + """Delivery side (``group_notification_service``): who actually gets mailed. + + The email plugin is mocked, so these pin recipient selection — the live + re-read on a grant, the ``revoked_at`` cutoff, org scoping and retained + access — not template or transport behavior. The enqueue side is covered in + ``test_share_notification_dispatch`` (unit tier, no DB). + """ + + def setUp(self) -> None: + super().setUp() + self.service = Mock() + patcher = patch( + "tenant_account_v2.group_notification_service.notification_plugin", + {"service_class": Mock(return_value=self.service)}, + ) + patcher.start() + self.addCleanup(patcher.stop) + + def _send( + self, + *, + group_ids: list[int], + share_action: str = ShareAction.SHARED.value, + revoked_at=None, + ) -> None: + send_resource_shared( + organization=self.org, + group_ids=group_ids, + actor_id=self.owner.pk, + resource_kind="workflow", + resource_id=str(self.workflow.pk), + share_action=share_action, + revoked_at=revoked_at, + ) + + def _mailed(self) -> list[tuple[str, list[str]]]: + """``(group_name, sorted recipient emails)`` per email sent, in order.""" + return [ + (call.kwargs["group_name"], sorted(u.email for u in call.kwargs["shared_to"])) + for call in self.service.send_group_resource_shared_notification.call_args_list + ] + + def test_grant_mails_current_group_members(self) -> None: + set_resource_share_groups(self.workflow, [self.group.id]) + self._send(group_ids=[self.group.id]) + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + + def test_grant_dropped_when_share_revoked_before_delivery(self) -> None: + # The queue can lag; announcing access the group no longer holds would + # disclose the resource name and id to members who cannot reach it. + set_resource_share_groups(self.workflow, [self.group.id]) + set_resource_share_groups(self.workflow, []) + self._send(group_ids=[self.group.id]) + self.service.send_group_resource_shared_notification.assert_not_called() + + def test_revoke_mails_members_although_the_share_row_is_gone(self) -> None: + # Mirror image of the check above: on a revoke the row is *expected* to + # be absent, so the live re-read must not suppress the mail. + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + kwargs = self.service.send_group_resource_shared_notification.call_args.kwargs + self.assertEqual(kwargs["share_action"], "revoked") + self.assertEqual(kwargs["resource_type"], "workflow") + self.assertEqual(kwargs["resource_name"], "wf-1") + + def test_group_from_another_org_is_never_mailed(self) -> None: + other_org = Organization.objects.create( + name="org-b", display_name="Org B", organization_id="org-b" + ) + foreign_group = OrganizationGroup.objects.create( + organization=other_org, name="Foreign", created_by=self.owner + ) + for action in (ShareAction.SHARED.value, ShareAction.REVOKED.value): + self._send(group_ids=[foreign_group.id], share_action=action) + self.service.send_group_resource_shared_notification.assert_not_called() + + def test_revoke_skips_members_who_joined_after_the_cutoff(self) -> None: + revoked_at = timezone.now() + latecomer = GroupMembership.objects.create(group=self.group, user=self.outsider) + # ``created_at`` is auto-set on save, so move it past the cutoff directly. + GroupMembership.objects.filter(pk=latecomer.pk).update( + created_at=revoked_at + timedelta(minutes=1) + ) + self._send( + group_ids=[self.group.id], + share_action=ShareAction.REVOKED.value, + revoked_at=revoked_at, + ) + # ``outsider`` never held access through this group, so no revoke notice. + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + + def test_revoke_skips_members_who_keep_access_another_way(self) -> None: + _add_viewers(self.workflow, self.member) + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + # Nothing was lost — a direct VIEWER row still reaches the resource. + self.service.send_group_resource_shared_notification.assert_not_called() + + def test_membership_change_mails_only_the_changed_users(self) -> None: + send_membership_changed( + organization=self.org, + group_id=self.group.id, + actor_id=self.owner.pk, + membership_action=MembershipAction.ADDED.value, + user_ids=[self.outsider.pk], + ) + kwargs = self.service.send_group_membership_notification.call_args.kwargs + self.assertEqual(kwargs["group_name"], "Team") + self.assertEqual(kwargs["membership_action"], "added") + self.assertEqual( + [u.email for u in kwargs["recipients"]], ["outsider@example.com"] + ) From be6f99983bc516db5c4a5ab146f645071d34fcef Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Thu, 10 Sep 2026 16:45:59 +0530 Subject: [PATCH 11/28] UN-3494 [MISC] Drop the group-notification feature flag The flag existed only because the PG queue it dispatches onto was itself behind pg_queue_enabled. UN-4046 removed that flag and made PG the only transport, and this feature merges in one shot rather than landing in main a piece at a time, so there is nothing left for a flag to buy. Removes _feature_enabled, GROUP_NOTIFICATION_FLAG_KEY and the FLIPT_SERVICE_AVAILABLE pre-check. The two dispatch guards keep their real checks -- a resolvable org, and a resource kind the email plugin has a type for. The revoke cutoff stays: the queue can still lag, so a member who joins after the access was taken away must not be mailed about it. Its comment no longer cites a Flipt round-trip, and the regression test that guarded that specific window goes with it -- nothing slow sits between the stamp and its use now. Verified: 45 tests pass across the three notification modules. Mutating both surviving guards to `if False:` fails exactly test_unknown_resource_kind_ skips_dispatch and test_missing_organization_skips_dispatch, so what remains is covered rather than merely present. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01LfrKHgSfDQxaXwJWrMxUWG --- .../tenant_account_v2/share_notifications.py | 54 +++------------- .../test_share_notification_dispatch.py | 64 +++++-------------- 2 files changed, 23 insertions(+), 95 deletions(-) diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 7616c4cd14..59354d0670 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -10,18 +10,15 @@ there is (UN-4046) — the ``notifications`` queue the notification consumer polls. -The group feature sits behind its own Flipt flag, evaluated once here at -enqueue: a blind Flipt, a missing org, or any dispatch error means no -notification, never a broken share. Turning the flag off stops new enqueues; a -message already queued still delivers. Direct-user share and revoke mail -(``ResourceShareManagementMixin``) is not on this flag — like the co-owner mail -it reuses, it is gated only by the cloud ``ENABLE_EMAIL_NOTIFICATIONS`` setting. +A missing org or any dispatch error means no notification, never a broken +share. Like the direct-user mail in ``ResourceShareManagementMixin`` and the +co-owner mail it reuses, sending is bounded only by the cloud +``ENABLE_EMAIL_NOTIFICATIONS`` setting. """ from __future__ import annotations import logging -import os from collections.abc import Iterable from enum import StrEnum from typing import TYPE_CHECKING, Any @@ -29,7 +26,6 @@ from django.utils import timezone from tenant_account_v2.shareable_resources import kind_for_instance -from unstract.flags.feature_flag import check_feature_flag_status if TYPE_CHECKING: from account_v2.models import User @@ -38,9 +34,6 @@ logger = logging.getLogger(__name__) -# Rollout flag for the whole feature — one constant so a grep finds every gate. -GROUP_NOTIFICATION_FLAG_KEY = "group_sharing_notifications_enabled" - NOTIFY_RESOURCE_SHARED_TASK = "notify_resource_shared_with_group" NOTIFY_MEMBERSHIP_CHANGED_TASK = "notify_group_membership_changed" @@ -103,14 +96,14 @@ def _notify_group_share( group_ids = sorted(group.pk for group in groups) if not group_ids: return - # Stamped before the Flipt round-trip below: a member joining inside that - # window would be mailed a revocation for access they never held. + # Stamped at the moment of the revoke, not at delivery: a member joining + # after it would otherwise be mailed about access they never held. revoked_at = ( timezone.now().isoformat() if share_action is ShareAction.REVOKED else None ) organization_id = _organization_slug(resource) kind = kind_for_instance(resource) - if not organization_id or kind is None or not _feature_enabled(organization_id): + if not organization_id or kind is None: return kwargs: dict[str, Any] = { "group_ids": group_ids, @@ -146,7 +139,7 @@ def notify_group_membership_changed( if not recipients: return organization_id = _organization_slug(group) - if not organization_id or not _feature_enabled(organization_id): + if not organization_id: return _dispatch_quietly( task_name=NOTIFY_MEMBERSHIP_CHANGED_TASK, @@ -161,37 +154,6 @@ def notify_group_membership_changed( ) -def _feature_enabled(organization_id: str) -> bool: - """Whether group-sharing notifications are on for this org. Fails closed.""" - # Parse exactly as FliptClient does (``.lower()``, no ``.strip()``) so the - # two can never disagree on a value like " true". - if os.environ.get("FLIPT_SERVICE_AVAILABLE", "false").lower() != "true": - logger.warning( - "group-notification: FLIPT_SERVICE_AVAILABLE != true (Flipt blind) " - "for org %s; skipping", - organization_id, - ) - return False - try: - enabled = bool( - check_feature_flag_status( - flag_key=GROUP_NOTIFICATION_FLAG_KEY, - entity_id=organization_id, - context={"organization_id": organization_id}, - ) - ) - except Exception: - logger.warning( - "group-notification: Flipt evaluation failed for org %s; skipping", - organization_id, - exc_info=True, - ) - return False - if not enabled: - logger.info("group-notification: flag off for org %s; skipping", organization_id) - return enabled - - def _organization_slug(obj: Any) -> str | None: """The owning org's string identifier (``Organization.organization_id``). diff --git a/backend/tenant_account_v2/test_share_notification_dispatch.py b/backend/tenant_account_v2/test_share_notification_dispatch.py index 672f29b423..4dc5063a16 100644 --- a/backend/tenant_account_v2/test_share_notification_dispatch.py +++ b/backend/tenant_account_v2/test_share_notification_dispatch.py @@ -2,16 +2,15 @@ ``share_notifications`` runs inside the user's share request: it builds the task payload and hands it to the transport. Nothing here touches the ORM or sends -mail, so the module is patched at its three seams — ``kind_for_instance``, -``_feature_enabled`` and ``_dispatch`` — and these run in the rig's unit tier -with no Postgres. The transport itself is covered by ``pg_queue.tests``; the +mail, so the module is patched at its two seams — ``kind_for_instance`` and +``_dispatch`` — and these run in the rig's unit tier with no Postgres. The transport itself is covered by ``pg_queue.tests``; the delivery side by ``ResourceShareNotificationTests`` in ``tenant_account_v2.tests``. """ from __future__ import annotations from contextlib import contextmanager -from datetime import UTC, datetime, timedelta +from datetime import datetime from types import SimpleNamespace from unittest.mock import patch @@ -28,19 +27,18 @@ def _group(pk: int) -> SimpleNamespace: @contextmanager -def _seams(*, enabled: bool = True, dispatch_raises: Exception | None = None): - """Patch the module's three outbound seams; yield the flag + dispatch mocks.""" +def _seams(*, dispatch_raises: Exception | None = None): + """Patch the module's two outbound seams; yield the dispatch mock.""" with ( patch.object(sn, "kind_for_instance", return_value="workflow"), - patch.object(sn, "_feature_enabled", return_value=enabled) as flag, patch.object(sn, "_dispatch", side_effect=dispatch_raises) as dispatch, ): - yield flag, dispatch + yield dispatch class TestNotifyResourceGroupShareChanged: def test_grant_dispatches_shared_payload_without_timestamp(self): - with _seams() as (_, dispatch): + with _seams() as dispatch: sn.notify_resource_group_share_changed( resource=_RESOURCE, added=[_group(5), _group(2)], removed=[], actor=_ACTOR ) @@ -60,7 +58,7 @@ def test_grant_dispatches_shared_payload_without_timestamp(self): assert "revoked_at" not in call["kwargs"] def test_revoke_dispatches_revoked_payload_with_timestamp(self): - with _seams() as (_, dispatch): + with _seams() as dispatch: sn.notify_resource_group_share_changed( resource=_RESOURCE, added=[], removed=[_group(3)], actor=_ACTOR ) @@ -70,30 +68,8 @@ def test_revoke_dispatches_revoked_payload_with_timestamp(self): # ISO-8601 string, not a datetime — the payload is JSON-serialized. datetime.fromisoformat(payload["revoked_at"]) - def test_revoked_at_is_stamped_before_the_flipt_round_trip(self): - # Regression (PR #2224): the stamp used to sit below ``_feature_enabled``, - # whose Flipt call is a network round-trip. Someone joining the group - # inside that window is mailed a revocation for access never held. - clock = [datetime(2026, 1, 1, 12, 0, tzinfo=UTC)] - - def _flipt(_org: str) -> bool: - clock[0] += timedelta(seconds=5) # stand-in for the Flipt round-trip - return True - - with ( - patch.object(sn.timezone, "now", side_effect=lambda: clock[0]), - patch.object(sn, "kind_for_instance", return_value="workflow"), - patch.object(sn, "_feature_enabled", side_effect=_flipt), - patch.object(sn, "_dispatch") as dispatch, - ): - sn.notify_resource_group_share_changed( - resource=_RESOURCE, added=[], removed=[_group(3)], actor=_ACTOR - ) - revoked_at = dispatch.call_args.kwargs["kwargs"]["revoked_at"] - assert revoked_at == datetime(2026, 1, 1, 12, 0, tzinfo=UTC).isoformat() - def test_grant_and_revoke_dispatch_independently(self): - with _seams() as (_, dispatch): + with _seams() as dispatch: sn.notify_resource_group_share_changed( resource=_RESOURCE, added=[_group(1)], removed=[_group(2)], actor=_ACTOR ) @@ -101,25 +77,16 @@ def test_grant_and_revoke_dispatch_independently(self): actions = [c.kwargs["kwargs"]["share_action"] for c in dispatch.call_args_list] assert actions == ["shared", "revoked"] - def test_no_groups_skips_before_the_flag_check(self): - with _seams() as (flag, dispatch): + def test_no_groups_skips_dispatch(self): + with _seams() as dispatch: sn.notify_resource_group_share_changed( resource=_RESOURCE, added=[], removed=[], actor=_ACTOR ) dispatch.assert_not_called() - flag.assert_not_called() # no Flipt call for a no-op share - - def test_flag_off_skips_dispatch(self): - with _seams(enabled=False) as (_, dispatch): - sn.notify_resource_group_share_changed( - resource=_RESOURCE, added=[_group(1)], removed=[], actor=_ACTOR - ) - dispatch.assert_not_called() def test_unknown_resource_kind_skips_dispatch(self): with ( patch.object(sn, "kind_for_instance", return_value=None), - patch.object(sn, "_feature_enabled", return_value=True), patch.object(sn, "_dispatch") as dispatch, ): sn.notify_resource_group_share_changed( @@ -129,7 +96,7 @@ def test_unknown_resource_kind_skips_dispatch(self): def test_missing_organization_skips_dispatch(self): orphan = SimpleNamespace(pk="wf-1", organization=None) - with _seams() as (_, dispatch): + with _seams() as dispatch: sn.notify_resource_group_share_changed( resource=orphan, added=[_group(1)], removed=[], actor=_ACTOR ) @@ -145,7 +112,7 @@ def test_dispatch_failure_never_reaches_the_caller(self): class TestNotifyGroupMembershipChanged: def test_membership_change_dispatches_user_ids_in_payload(self): - with _seams() as (_, dispatch): + with _seams() as dispatch: sn.notify_group_membership_changed( group=SimpleNamespace( pk=9, organization=SimpleNamespace(organization_id="org-a") @@ -166,8 +133,8 @@ def test_membership_change_dispatches_user_ids_in_payload(self): "organization_id": "org-a", } - def test_no_users_skips_before_the_flag_check(self): - with _seams() as (flag, dispatch): + def test_no_users_skips_dispatch(self): + with _seams() as dispatch: sn.notify_group_membership_changed( group=SimpleNamespace( pk=9, organization=SimpleNamespace(organization_id="org-a") @@ -177,4 +144,3 @@ def test_no_users_skips_before_the_flag_check(self): actor=_ACTOR, ) dispatch.assert_not_called() - flag.assert_not_called() From a30d1ee7c5a563eaa88305f4e56766d210a8b7ae Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 15 Sep 2026 00:16:02 +0530 Subject: [PATCH 12/28] UN-3494 [FIX] Close review findings on the group-notification path F2 A plugin that fails to import logs only at DEBUG, so a cloud build missing sendgrid mails nobody while every layer reports success. Warn when email is switched on and the plugin is absent -- the one state that can only be a broken build. F4 _retained_user_ids hydrated every member of the org on a revoke of an org-shared resource, to conclude nobody lost access. Short-circuit it, the same guard _users_left_without_access already carries. F5 The enqueue guards returned silently on a missing org and on an unregistered resource kind, both defects rather than routine skips; and _groups_to_mail dropped groups with no count. F6 State the deploy ordering at the enqueue site: a consumer on the previous image cannot resolve these task names and DELETES the rows, with no dead-letter, since nothing here passes reply_key or on_error. F7 The at-least-once note claimed one duplicate email. A retry re-posts the whole payload and the backend mails group by group with no checkpoint, so the bound is 3 attempts times the consumer's cap. F8 VT is 300s and its justification described one POST per task. HTTPTransport(retries=2) retries the CONNECT, so worst case is ~274s -- measured, not inferred. F11 A failed add no longer falls through to the removals: the roster never grew, so removing could strip the very owner the swap was replacing. F12 refreshCoOwnerData returns a verdict instead of leaving four post-await sites to re-check the ref, only one of which did. A late apply could close whichever co-owner modal was open by then. F13 Restore the `applying` half of the body guard this branch dropped, so edits made mid-apply are not silently discarded by the re-seed; and correct the comment promising a retry surface the re-seed removes. F14 Seed the staged roster during render rather than in an effect, so the first frame of a new resource cannot show the previous resource's list. F15 Say why shared_to_org is not a notification axis. F16 share_action and revoked_at are always sent, so drop the defaults that would turn a renamed field into a revoke mailed as a share. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/resource_share_views.py | 3 ++ .../group_notification_service.py | 39 +++++++++++++++-- backend/tenant_account_v2/internal_views.py | 8 ++-- .../tenant_account_v2/share_notifications.py | 24 +++++++++++ docker/docker-compose.yaml | 3 +- .../co-owner-management/CoOwnerManagement.jsx | 18 +++++--- frontend/src/hooks/useCoOwnerManagement.jsx | 42 ++++++++++++++----- workers/notification/tasks.py | 12 +++++- 8 files changed, 123 insertions(+), 26 deletions(-) diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index fbc61de1b7..660af7dd91 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -145,6 +145,9 @@ def share(self, request: Request, pk: str | None = None) -> Response: # committed — the diffs read persisted state and can never announce a # share that rolled back. resource.refresh_from_db() + # Only the two per-recipient axes notify. ``shared_to_org`` is left out + # deliberately: a toggle has no recipient list short of the whole org, + # and it is read below as a reason someone KEPT access, not lost it. users_after = self._read_axis(resource, "shared_users") groups_after = self._read_axis(resource, "shared_groups") notify_resource_group_share_changed( diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index e577e9359f..1f6834e1de 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -17,6 +17,7 @@ from account_v2.models import Organization, User from django.apps import apps +from django.conf import settings from django.db.models import QuerySet from plugins import get_plugin @@ -98,6 +99,8 @@ def send_resource_shared( ) return retained = _retained_user_ids(shared.instance, share_action) + if retained is None: + return for group in _groups_to_mail(organization, group_ids, shared.instance, share_action): recipients = _group_recipients(organization, group, retained, revoked_at) logger.info( @@ -162,7 +165,16 @@ def send_membership_changed( def _service() -> Any | None: """The cloud email service, or ``None`` when the plugin is absent (OSS).""" if not notification_plugin: - logger.debug("group-notification: notification plugin unavailable, skipping") + # An absent plugin is normal in OSS. Absent while email is switched on + # can only be a broken build, and the plugin loader swallows the import + # error at DEBUG, so this is the only place it can surface. + if getattr(settings, "ENABLE_EMAIL_NOTIFICATIONS", False): + logger.warning( + "group-notification: email is enabled but the notification " + "plugin did not load — no mail is being sent" + ) + else: + logger.debug("group-notification: notification plugin unavailable, skipping") return None return notification_plugin["service_class"]() @@ -181,7 +193,7 @@ def _get_user(organization: Organization, user_id: int) -> User | None: return member.user if member else None -def _retained_user_ids(resource: Any, share_action: str) -> set[int]: +def _retained_user_ids(resource: Any, share_action: str) -> set[int] | None: """Users who still reach ``resource``; empty on the share direction. A revoked group's members may keep access through another group, a direct @@ -189,9 +201,22 @@ def _retained_user_ids(resource: Any, share_action: str) -> set[int]: and the revoke email also repoints their CTA at the dashboard. Owners sit outside ``compute_effective_members`` by design, so add them back: an owner in the revoked group has lost nothing. + + ``None`` means an org-wide share still covers everyone, so the caller skips + the fan-out entirely. """ if share_action != ShareAction.REVOKED.value: return set() + if getattr(resource, "shared_to_org", False): + # Org-wide share still covers everyone — nobody lost access, and + # answering it via ``compute_effective_members`` would hydrate every + # member of the org to say so (same guard as the direct-share path). + logger.info( + "group-notification: revoke on an org-shared resource %s — " + "nobody lost access, no mail", + resource.pk, + ) + return None from tenant_account_v2.sharing_helpers import compute_effective_members return {member["user_id"] for member in compute_effective_members(resource)} | { @@ -219,7 +244,15 @@ def _groups_to_mail( from tenant_account_v2.sharing_helpers import get_resource_share_groups live = {group.pk for group in get_resource_share_groups(resource)} - return [group for group in groups if group.pk in live] + to_mail = [group for group in groups if group.pk in live] + if len(to_mail) != len(groups): + logger.info( + "group-notification: dropped %d of %d groups " + "(access revoked or group gone since enqueue)", + len(groups) - len(to_mail), + len(groups), + ) + return to_mail def _group_recipients( diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index 47c292580f..0e10a2cf8c 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -37,12 +37,12 @@ class ResourceSharedWithGroupSerializer(serializers.Serializer): actor_id = serializers.IntegerField() resource_kind = serializers.CharField() resource_id = serializers.CharField() - share_action = serializers.ChoiceField( - choices=[a.value for a in ShareAction], default=ShareAction.SHARED.value - ) + # Both are required: the worker sends them on every call, so a default here + # would turn a renamed field into a revoke silently mailed as a share. + share_action = serializers.ChoiceField(choices=[a.value for a in ShareAction]) # Revoke only: members who joined after this are excluded from the mail. # Nullable because the worker sends the key on both directions. - revoked_at = serializers.DateTimeField(allow_null=True, default=None) + revoked_at = serializers.DateTimeField(allow_null=True) class GroupMembershipChangedSerializer(serializers.Serializer): diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 59354d0670..35fee89bfc 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -104,6 +104,18 @@ def _notify_group_share( organization_id = _organization_slug(resource) kind = kind_for_instance(resource) if not organization_id or kind is None: + # Neither is a routine skip: a shareable resource always resolves an + # org, and a kind the registry does not carry means the resource + # reached a share endpoint it was never registered for. + logger.warning( + "group-notification: skipping %s share for %s %s " + "(organization=%s kind=%s)", + share_action, + type(resource).__name__, + resource.pk, + organization_id, + kind, + ) return kwargs: dict[str, Any] = { "group_ids": group_ids, @@ -140,6 +152,11 @@ def notify_group_membership_changed( return organization_id = _organization_slug(group) if not organization_id: + logger.warning( + "group-notification: skipping membership change for group %s " + "(no resolvable organization)", + group.pk, + ) return _dispatch_quietly( task_name=NOTIFY_MEMBERSHIP_CHANGED_TASK, @@ -195,6 +212,13 @@ def _dispatch( kwargs: dict[str, Any], organization_id: str, ) -> None: + """Enqueue on the PG queue. + + Deploy the notification worker at or before the backend: the consumer polls + the queue by name, so a pod still on the previous image claims these rows, + cannot resolve the task, and DELETES them (no dead-letter — nothing here + passes ``reply_key`` or ``on_error``). + """ # Lazy import — ``pg_queue`` is heavier than this leaf module and importing # it at load time risks a cycle during Django app loading. from pg_queue.producer import enqueue_task diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index e89ca0e19d..0c82e7fe1e 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -589,7 +589,8 @@ services: - WORKER_PG_QUEUE_CONSUMER_CONCURRENCY=${PG_NOTIFICATION_CONCURRENCY:-2} # Log publisher — see the note on the backend service (UN-4046). - LOG_TRANSPORT=redis - # One outbound HTTP POST per task. VT sits above a slow subscriber so a + # Up to 3 outbound HTTP POSTs per task (group-notification retries, ~274s + # worst case). VT sits above a slow subscriber so a # sibling replica cannot re-claim mid-POST and double-deliver; health-stale # above that. Mirrors the chart's workerPgNotification bounds. - WORKER_PG_QUEUE_CONSUMER_VT_SECONDS=${PG_NOTIFICATION_VT_SECONDS:-300} diff --git a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx index c75da57652..280f8cc030 100644 --- a/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx +++ b/frontend/src/components/widgets/co-owner-management/CoOwnerManagement.jsx @@ -1,6 +1,6 @@ import { CircleHelp, Trash2, User } from "lucide-react"; import PropTypes from "prop-types"; -import { useEffect, useMemo, useState } from "react"; +import { useMemo, useState } from "react"; import { Button } from "@/components/ui/shims/antd-button"; import { Select } from "@/components/ui/shims/antd-inputs"; import { Avatar } from "@/components/ui/shims/antd-leaves"; @@ -32,9 +32,15 @@ function CoOwnerManagement({ // after an Apply. Doubles as the reset — most hosts leave this modal mounted, // and the hook can close it without ``handleCancel`` (404 / fetch-error), so // staged edits must not leak into the next resource. - useEffect(() => { + // + // Done during render, not in an effect: an effect commits after the one that + // reveals the new roster, so the first frame of a new resource would render + // the previous resource's staged list (and enable Apply on that diff). + const [seededFrom, setSeededFrom] = useState(null); + if (seededFrom !== ownersList) { + setSeededFrom(ownersList); setSelectedOwners(ownersList); - }, [ownersList]); + } const selectedIds = useMemo( () => new Set(selectedOwners.map((u) => u?.id?.toString())), @@ -79,8 +85,8 @@ function CoOwnerManagement({ } setApplying(true); try { - // Close only on a clean apply; a partial failure keeps the modal open so - // the user can see what was rejected and retry. + // Close only on a clean apply. A partial failure keeps the modal open on + // the refreshed server roster, with the alert naming who was rejected. if (await onApplyCoOwners(resourceId, { addUsers, removeUsers })) { setOpen(false); } @@ -111,7 +117,7 @@ function CoOwnerManagement({ closable={true} className="co-owner-modal" > - {loading ? ( + {loading || applying ? ( ) : ( <> diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index ed00733da7..d2399b13cc 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -54,15 +54,21 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // branch) after the user has moved to a different resource. Mutation // callers pass the token captured BEFORE their POSTs so a modal switch // during the mutation itself is caught too, not just one mid-refresh. - // Returns true when the resource turned out to be gone, so the caller can - // leave that alert standing instead of overwriting it with its own. + // Returns the verdict rather than leaving each caller to re-derive it: + // "stale" (superseded — touch no shared state), "gone" (404, alert already + // raised), "ok" (proceed). async (resourceId, requestId = latestRequestRef.current) => { try { const res = await service.getSharedUsers(resourceId); - if (latestRequestRef.current !== requestId) return; + if (latestRequestRef.current !== requestId) { + return "stale"; + } setCoOwnerData({ coOwners: res.data?.co_owners || [] }); + return "ok"; } catch (err) { - if (latestRequestRef.current !== requestId) return; + if (latestRequestRef.current !== requestId) { + return "stale"; + } if (err?.response?.status === 404) { setCoOwnerOpen(false); onListRefresh?.(); @@ -71,11 +77,12 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { content: "This resource is no longer accessible. It may have been removed or your access has been revoked.", }); - return true; + return "gone"; } setAlertDetails( handleException(err, "Unable to refresh co-owner data"), ); + return "ok"; } }, [service, onListRefresh, setAlertDetails, handleException], @@ -95,7 +102,9 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { service.getSharedUsers(resourceId), ]); - if (latestRequestRef.current !== requestId) return; + if (latestRequestRef.current !== requestId) { + return; + } const userList = usersResponse?.data?.members?.map((member) => ({ @@ -108,7 +117,9 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { coOwners: sharedUsersResponse.data?.co_owners || [], }); } catch (err) { - if (latestRequestRef.current !== requestId) return; + if (latestRequestRef.current !== requestId) { + return; + } setAlertDetails( handleException(err, "Unable to fetch co-owner information"), ); @@ -142,10 +153,21 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // Adds first: the backend rejects removing the last owner, so a one-shot // owner swap has to grow the roster before it shrinks it. await run(addUsers, (id) => service.addCoOwner(resourceId, id)); - await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); + if (failed.length) { + // The roster never grew, so removing now can strip the very owner the + // swap was meant to replace. Report them rather than attempt them. + failed.push(...removeUsers); + } else { + await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); + } // Reconverge the modal on true server state regardless of partial outcome. - const gone = await refreshCoOwnerData(resourceId, requestId); - if (gone) { + const outcome = await refreshCoOwnerData(resourceId, requestId); + if (outcome === "stale") { + // The user has opened another resource since Apply. Closing the modal + // or alerting now would hit that one instead of this. + return false; + } + if (outcome === "gone") { // The refresh already closed the modal, refreshed the list and raised // its own alert — an apply summary on top of it would only mislead. return true; diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 7ef9bc4c32..9c850b30b1 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -510,6 +510,11 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # Retries for a transient backend problem (restart, 5xx). Kept inside the task # so a brief blip is absorbed here rather than costing a full lease-expiry # redelivery (minutes) plus one of the consumer's bounded attempts. +# +# These two bound the task's wall time, which must stay under the consumer's +# visibility timeout (300s): ``HTTPTransport(retries=2)`` retries the CONNECT, +# so one post is up to 3 x 30s, and three attempts plus the sleeps reach ~274s. +# Raising either constant, or the per-post timeout, overruns the VT. _GROUP_NOTIFICATION_ATTEMPTS = 3 _GROUP_NOTIFICATION_RETRY_DELAY = 2.0 @@ -524,8 +529,11 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) A sub-500 response ends the in-process attempts — a rejected payload will be rejected again. Delivery is at-least-once: a response lost after the - backend already sent re-posts the same payload, and the send path writes - nothing, so the only effect is a duplicate email. + backend already sent re-posts the *whole* payload, and the backend mails + group by group with no checkpoint, so a failure partway through the fan-out + re-mails the groups that already succeeded. The send path writes nothing, so + duplicate email is the only effect — but the bound is these 3 attempts times + the consumer's attempt cap, not one. """ base_url = os.getenv("INTERNAL_API_BASE_URL") api_key = os.getenv("INTERNAL_SERVICE_API_KEY") From 3c78994cde4113ab4060cb2cf3ffde4af1266594 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 15 Sep 2026 08:44:11 +0530 Subject: [PATCH 13/28] UN-3494 [FIX] Correct the retry budget, the share_action default and the apply verdicts Verification of the previous commit found these in its own new lines. - The ~274s figure omitted httpcore's 0.5s/1.0s connect backoff; the real ceiling is ~278s against a 300s VT. The closing sentence claimed raising either constant overruns the budget, which is false for a modest bump of the retry delay -- exactly the wrong thing to leave a maintainer reasoning from. - The serializer comment said a default could not turn a renamed field into a revoke mailed as a share, but the worker task defaulted share_action to "shared" one hop upstream, where the serializer cannot see it. The task now requires it. revoked_at keeps its default: the producer omits it on the share direction deliberately. - The group-drop log offered "or group gone since enqueue" as a cause, but _groups_in_org removes deleted and out-of-org groups before the count, so that alternative can never be the one reported. - refreshCoOwnerData returned "ok" after a refresh that failed, so the caller stacked a success summary on a standing error alert and closed the modal on a roster it could not verify. It now returns "error". - The stale verdict also skipped onListRefresh, but the mutations had already landed and the list is page-scoped, not modal-scoped. - "Failed for:" named removals that were deliberately never attempted after an addition failed. Now "Not applied for:", which is true of both. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- .../group_notification_service.py | 2 +- backend/tenant_account_v2/internal_views.py | 4 +-- frontend/src/hooks/useCoOwnerManagement.jsx | 30 +++++++++++-------- workers/notification/tasks.py | 7 +++-- 4 files changed, 25 insertions(+), 18 deletions(-) diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 1f6834e1de..480164b966 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -248,7 +248,7 @@ def _groups_to_mail( if len(to_mail) != len(groups): logger.info( "group-notification: dropped %d of %d groups " - "(access revoked or group gone since enqueue)", + "(access revoked since enqueue)", len(groups) - len(to_mail), len(groups), ) diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index 0e10a2cf8c..e884cf27e3 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -37,8 +37,8 @@ class ResourceSharedWithGroupSerializer(serializers.Serializer): actor_id = serializers.IntegerField() resource_kind = serializers.CharField() resource_id = serializers.CharField() - # Both are required: the worker sends them on every call, so a default here - # would turn a renamed field into a revoke silently mailed as a share. + # Required, and the worker task takes no default for it either: a default + # on both sides would turn a dropped field into a revoke mailed as a share. share_action = serializers.ChoiceField(choices=[a.value for a in ShareAction]) # Revoke only: members who joined after this are excluded from the mail. # Nullable because the worker sends the key on both directions. diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index d2399b13cc..a6e579aea6 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -35,7 +35,12 @@ function buildApplyAlert( return { type: "success", content: summary }; } const failedNames = failed.map((user) => user?.email || user?.id).join(", "); - return { type: "warning", content: `${summary}. Failed for: ${failedNames}` }; + // "Not applied" rather than "Failed": this list also carries removals that + // were deliberately skipped because an addition failed first. + return { + type: "warning", + content: `${summary}. Not applied for: ${failedNames}`, + }; } function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { @@ -56,7 +61,8 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // during the mutation itself is caught too, not just one mid-refresh. // Returns the verdict rather than leaving each caller to re-derive it: // "stale" (superseded — touch no shared state), "gone" (404, alert already - // raised), "ok" (proceed). + // raised), "error" (refresh failed, alert already raised, roster + // unverified), "ok" (proceed). async (resourceId, requestId = latestRequestRef.current) => { try { const res = await service.getSharedUsers(resourceId); @@ -82,7 +88,7 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { setAlertDetails( handleException(err, "Unable to refresh co-owner data"), ); - return "ok"; + return "error"; } }, [service, onListRefresh, setAlertDetails, handleException], @@ -162,17 +168,17 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { } // Reconverge the modal on true server state regardless of partial outcome. const outcome = await refreshCoOwnerData(resourceId, requestId); - if (outcome === "stale") { - // The user has opened another resource since Apply. Closing the modal - // or alerting now would hit that one instead of this. - return false; + if (outcome !== "gone") { + // The mutations landed whatever the refresh said, and the list is + // page-scoped rather than modal-scoped. ("gone" already refreshed it.) + onListRefresh?.(); } - if (outcome === "gone") { - // The refresh already closed the modal, refreshed the list and raised - // its own alert — an apply summary on top of it would only mislead. - return true; + if (outcome !== "ok") { + // "stale": the user has opened another resource, so closing or + // alerting would hit that one. "gone"/"error": an alert is already + // standing, and a summary on top of it would only mislead. + return outcome === "gone"; } - onListRefresh?.(); setAlertDetails( buildApplyAlert( addUsers, diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 9c850b30b1..eccc4dc4d4 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -513,8 +513,9 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # # These two bound the task's wall time, which must stay under the consumer's # visibility timeout (300s): ``HTTPTransport(retries=2)`` retries the CONNECT, -# so one post is up to 3 x 30s, and three attempts plus the sleeps reach ~274s. -# Raising either constant, or the per-post timeout, overruns the VT. +# so one post is up to 3 x 30s; three attempts, the sleeps, and httpcore's own +# 0.5s/1.0s connect backoff reach ~278s. Check that budget before raising +# either constant or the per-post timeout. _GROUP_NOTIFICATION_ATTEMPTS = 3 _GROUP_NOTIFICATION_RETRY_DELAY = 2.0 @@ -581,7 +582,7 @@ def notify_resource_shared_with_group( resource_kind: str, resource_id: str, organization_id: str, - share_action: str = "shared", + share_action: str, revoked_at: str | None = None, ) -> None: """Email every current member of the groups whose access just changed. From 0f82e0d3c16e71329ff32e864cb3653250db9458 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 15 Sep 2026 19:27:43 +0530 Subject: [PATCH 14/28] UN-3494 [FIX] Bound the notification POST per phase, and close the review's comment and observability gaps N3/N4 The post could outlive the queue's visibility timeout. httpx has no whole-request timeout: a scalar `timeout=30.0` applies to connect, write and read SEPARATELY, and `HTTPTransport(retries=2)` retries only the CONNECT phase, inside one post, stacking its own timeouts underneath. Worst case was connect-fail 30 + 30, backoff 0.5, connect OK, write 30, read 30 -- about 120s per post, ~365s for the task, over both VT (300s, a sibling re-claims mid-fan-out) and health-stale (360s, the pod is restarted mid-task). Earlier derivations missed this because they only measured the all-connect-fail case. Now bounded per phase, with the task's own loop as the only retry: 5+10+30+5 = 50s a post, 154s for the task. A read or write timeout also ends the in-process attempts, because the request reached the backend, the backend does not stop when we disconnect, and the send path has no checkpoint -- re-posting re-mails every group that already succeeded. N1 The module's "failure contract" paragraph said a deleted resource was the ONE deliberate 200-on-failure. The cloud revert made a failed SendGrid send a second one. Deleted rather than rewritten: both real sites state the carve-out correctly where it happens. N2 An unregistered resource kind is not necessarily a client bug -- the share endpoint accepts `shared_groups` for any host viewset. N5 Neither failure on this path carried a `metric=` prefix, which is what the metrics pipeline scrapes, while every sibling webhook path has one. Added to the enqueue swallow and the worker's terminal raise. N6 `send_resource_shared` kept a `share_action` default -- exactly the trap its own serializer comment documents holding the line against. R1/R2 One derivation of the retry budget, in the code. The two config comments point at it instead of restating a number that was wrong in three different ways across three files. R3 The "error" verdict left the modal open on a stale roster with the staged diff intact, so a second Apply re-posted mutations the backend had already accepted and reported every one as a failure. R4 The verdict legend contradicted its only caller on "stale". Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- .../group_notification_service.py | 2 +- backend/tenant_account_v2/internal_views.py | 5 +-- .../tenant_account_v2/share_notifications.py | 7 ++-- docker/docker-compose.yaml | 5 ++- frontend/src/hooks/useCoOwnerManagement.jsx | 25 ++++++++---- workers/notification/tasks.py | 38 +++++++++++++++---- 6 files changed, 57 insertions(+), 25 deletions(-) diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 480164b966..bed52f9667 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -74,7 +74,7 @@ def send_resource_shared( actor_id: int, resource_kind: str, resource_id: str, - share_action: str = ShareAction.SHARED.value, + share_action: str, revoked_at: datetime | None = None, ) -> None: """Mail every current member of each group whose resource access changed. diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index e884cf27e3..5cc38b52be 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -5,9 +5,8 @@ step of the send — group expansion, org re-validation, resource lookup, the email plugin — needs it. -Failure contract: **any** unhandled problem must surface as non-2xx so the -worker retries. The one deliberate exception is a resource that no longer -exists, which returns 200 — retrying that can only fail again. +An unhandled problem surfaces as non-2xx so the worker retries. Cases that a +retry cannot help are answered 200 at the site that recognises them. """ import logging diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 35fee89bfc..c4565695e2 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -105,8 +105,9 @@ def _notify_group_share( kind = kind_for_instance(resource) if not organization_id or kind is None: # Neither is a routine skip: a shareable resource always resolves an - # org, and a kind the registry does not carry means the resource - # reached a share endpoint it was never registered for. + # org, and the share endpoint accepts ``shared_groups`` for any host + # viewset, so an unregistered kind means a group share landed on a + # resource this feature cannot mail about. logger.warning( "group-notification: skipping %s share for %s %s " "(organization=%s kind=%s)", @@ -200,7 +201,7 @@ def _dispatch_quietly( ) except Exception: logger.exception( - "group-notification: failed to dispatch %s for org %s", + "metric=group_notification_enqueue_failed_total task=%s org_id=%s", task_name, organization_id, ) diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index 0c82e7fe1e..845f646ce0 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -589,8 +589,9 @@ services: - WORKER_PG_QUEUE_CONSUMER_CONCURRENCY=${PG_NOTIFICATION_CONCURRENCY:-2} # Log publisher — see the note on the backend service (UN-4046). - LOG_TRANSPORT=redis - # Up to 3 outbound HTTP POSTs per task (group-notification retries, ~274s - # worst case). VT sits above a slow subscriber so a + # Several outbound HTTP POSTs per task -- the group-notification retry + # budget is derived at ``_GROUP_NOTIFICATION_ATTEMPTS`` in + # workers/notification/tasks.py. VT sits above a slow subscriber so a # sibling replica cannot re-claim mid-POST and double-deliver; health-stale # above that. Mirrors the chart's workerPgNotification bounds. - WORKER_PG_QUEUE_CONSUMER_VT_SECONDS=${PG_NOTIFICATION_VT_SECONDS:-300} diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index a6e579aea6..9bd8a7c980 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -60,9 +60,9 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // callers pass the token captured BEFORE their POSTs so a modal switch // during the mutation itself is caught too, not just one mid-refresh. // Returns the verdict rather than leaving each caller to re-derive it: - // "stale" (superseded — touch no shared state), "gone" (404, alert already - // raised), "error" (refresh failed, alert already raised, roster - // unverified), "ok" (proceed). + // "stale" (a later modal superseded this one), "gone" (404), "error" + // (refresh failed; roster unverified), "ok". "gone" and "error" have both + // already raised their own alert. async (resourceId, requestId = latestRequestRef.current) => { try { const res = await service.getSharedUsers(resourceId); @@ -173,12 +173,21 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { // page-scoped rather than modal-scoped. ("gone" already refreshed it.) onListRefresh?.(); } - if (outcome !== "ok") { - // "stale": the user has opened another resource, so closing or - // alerting would hit that one. "gone"/"error": an alert is already - // standing, and a summary on top of it would only mislead. - return outcome === "gone"; + if (outcome === "stale") { + // The user has opened another resource, so closing this modal or + // alerting would hit that one instead. + return false; } + if (outcome === "gone") { + // The refresh closed the modal and raised its own alert; a summary on + // top of it would only mislead. + return true; + } + // "ok" and "error" both fall through: the mutations landed either way and + // their outcomes are known without the refresh. Staying silent on "error" + // would leave the modal open on a stale roster whose staged diff still + // looks unapplied, and a second Apply would re-post mutations the backend + // has already accepted -- reporting every one as a failure. setAlertDetails( buildApplyAlert( addUsers, diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index eccc4dc4d4..8556e9edf8 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -510,14 +510,16 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # Retries for a transient backend problem (restart, 5xx). Kept inside the task # so a brief blip is absorbed here rather than costing a full lease-expiry # redelivery (minutes) plus one of the consumer's bounded attempts. -# -# These two bound the task's wall time, which must stay under the consumer's -# visibility timeout (300s): ``HTTPTransport(retries=2)`` retries the CONNECT, -# so one post is up to 3 x 30s; three attempts, the sleeps, and httpcore's own -# 0.5s/1.0s connect backoff reach ~278s. Check that budget before raising -# either constant or the per-post timeout. _GROUP_NOTIFICATION_ATTEMPTS = 3 _GROUP_NOTIFICATION_RETRY_DELAY = 2.0 +# Per-phase, because httpx has NO whole-request timeout: a scalar timeout is +# applied to connect, write and read separately, so each can spend it in full. +# The loop below is the only retry -- ``HTTPTransport(retries=2)`` would retry +# the connect phase *inside* one post, stacking its own timeouts under these. +# Sum x _GROUP_NOTIFICATION_ATTEMPTS must stay under the consumer's visibility +# timeout (300s) or a sibling re-claims the message mid-fan-out, and under +# health-stale (360s) or the pod is restarted mid-task. +_GROUP_NOTIFICATION_TIMEOUT = httpx.Timeout(connect=5.0, write=10.0, read=30.0, pool=5.0) def _post_group_notification(endpoint: str, organization_id: str, payload: dict) -> None: @@ -553,8 +555,22 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) last_error = "" for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): try: - with httpx.Client(transport=httpx.HTTPTransport(retries=2)) as client: - response = client.post(url, headers=headers, json=payload, timeout=30.0) + with httpx.Client() as client: + response = client.post( + url, + headers=headers, + json=payload, + timeout=_GROUP_NOTIFICATION_TIMEOUT, + ) + except (httpx.ReadTimeout, httpx.WriteTimeout) as e: + # The request reached the backend, and the backend does not stop + # when we disconnect. Its outcome is unknown, and the send path has + # no checkpoint, so re-posting re-mails every group that already + # succeeded. End the in-process attempts like a sub-500 does. + # (The queue can still redeliver; only an idempotency key on the + # handler would bound that, which this PR does not add.) + last_error = f"timeout_after_send={e!r}" + break except Exception as e: # noqa: BLE001 last_error = f"exception={e!r}" else: @@ -572,6 +588,12 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) last_error, ) time.sleep(_GROUP_NOTIFICATION_RETRY_DELAY) + logger.error( + "metric=group_notification_post_failed_total endpoint=%s org_id=%s error=%s", + endpoint, + organization_id, + last_error, + ) raise RuntimeError(f"Group notification {endpoint} failed: {last_error}") From 69474ee7c90aa6089dcb27fa17dd2e10c2157815 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 15 Sep 2026 19:48:24 +0530 Subject: [PATCH 15/28] UN-3494 [FIX] Widen the request-sent exception class, and delete the prose that keeps being wrong Verification of the previous commit found both. - The `break` encodes a rule -- the request left, the backend does not stop when we disconnect, so re-posting re-mails every group that already succeeded -- and enumerated two of its four members. `ReadError` and `RemoteProtocolError` satisfy it too, and a backend pod rollout mid-fan-out raises exactly those. They were falling through to the generic arm and buying two more full re-posts inside a single delivery. - Three comments have now been rewritten once per round and been wrong every time. The retry budget was ~274s, then ~278s, then a correct bound attached to a re-claim mechanism the renewable lease replaced (the consumer says so itself: VT is the drain bound, not the claim window). The failure contract was "the one deliberate 200", then "cases a retry cannot help" -- still false, because a transient SendGrid rejection is retryable and is answered 200 anyway. Each rewrite is a fresh set of claims and a fresh surface for the same defect, so these keep only what is checkable and drop the mechanism stories. The budget comment also no longer implies it covers DNS, which `connect` does not. - The co-owner fall-through comment claimed to close the re-Apply hole. It does not: on a partial failure the function still returns false and the staged diff is still computed against an unrefreshed roster. Narrowed to what the change actually does. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/tenant_account_v2/internal_views.py | 4 ++-- frontend/src/hooks/useCoOwnerManagement.jsx | 8 ++++---- workers/notification/tasks.py | 20 ++++++++++++-------- 3 files changed, 18 insertions(+), 14 deletions(-) diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index 5cc38b52be..d2cef14312 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -5,8 +5,8 @@ step of the send — group expansion, org re-validation, resource lookup, the email plugin — needs it. -An unhandled problem surfaces as non-2xx so the worker retries. Cases that a -retry cannot help are answered 200 at the site that recognises them. +A 200 does not imply the email was sent -- see the send path in +:mod:`tenant_account_v2.group_notification_service`. """ import logging diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index 9bd8a7c980..4a1b0978e5 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -184,10 +184,10 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { return true; } // "ok" and "error" both fall through: the mutations landed either way and - // their outcomes are known without the refresh. Staying silent on "error" - // would leave the modal open on a stale roster whose staged diff still - // looks unapplied, and a second Apply would re-post mutations the backend - // has already accepted -- reporting every one as a failure. + // their outcomes are known without the refresh, so a clean apply closes + // the modal rather than stranding it. A partial failure still leaves the + // staged diff computed against an unrefreshed roster -- clearing that is + // not done here. setAlertDetails( buildApplyAlert( addUsers, diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 8556e9edf8..f521850f15 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -512,13 +512,12 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # redelivery (minutes) plus one of the consumer's bounded attempts. _GROUP_NOTIFICATION_ATTEMPTS = 3 _GROUP_NOTIFICATION_RETRY_DELAY = 2.0 -# Per-phase, because httpx has NO whole-request timeout: a scalar timeout is -# applied to connect, write and read separately, so each can spend it in full. -# The loop below is the only retry -- ``HTTPTransport(retries=2)`` would retry -# the connect phase *inside* one post, stacking its own timeouts under these. -# Sum x _GROUP_NOTIFICATION_ATTEMPTS must stay under the consumer's visibility -# timeout (300s) or a sibling re-claims the message mid-fan-out, and under -# health-stale (360s) or the pod is restarted mid-task. +# Per-phase, because httpx has NO whole-request timeout: a scalar one is applied +# to connect, write and read separately. The loop below is the only retry -- +# transport-level retries would stack their own timeouts underneath these. +# Sum x _GROUP_NOTIFICATION_ATTEMPTS is the task's bound, and must stay under +# WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS (the heartbeat is frozen for the +# task's duration) and VT_SECONDS. Excludes DNS, which connect does not cover. _GROUP_NOTIFICATION_TIMEOUT = httpx.Timeout(connect=5.0, write=10.0, read=30.0, pool=5.0) @@ -562,7 +561,12 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) json=payload, timeout=_GROUP_NOTIFICATION_TIMEOUT, ) - except (httpx.ReadTimeout, httpx.WriteTimeout) as e: + except ( + httpx.ReadTimeout, + httpx.WriteTimeout, + httpx.ReadError, + httpx.RemoteProtocolError, + ) as e: # The request reached the backend, and the backend does not stop # when we disconnect. Its outcome is unknown, and the send path has # no checkpoint, so re-posting re-mails every group that already From 96beeb69629ad3b785a310dfa8dc8a933b207d83 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 15 Sep 2026 20:29:32 +0530 Subject: [PATCH 16/28] UN-3494 [TEST] Cover the worker leg and the recipient-selection guards A mutation pass over this branch found that large parts of the feature could be deleted with the whole suite still green. These close the ones that matter, and each was written by breaking the production line first and confirming the test fails. workers/tests/test_group_notification_post.py (18, unit tier) The worker leg had no test of any kind, and it is entirely failure-path logic. The branch that matters is which exceptions mean "the request reached the backend": those must not be re-posted, because the backend does not stop when we disconnect and the send path has no checkpoint, so a retry re-mails every group that already succeeded. Connect-side failures must still retry, because nothing happened. Also pins the attempt cap, the sub-500 break, the credential guard, both task payloads, and that the timeout is per-phase -- a scalar one silently triples the task's worst case and pushed it past the consumer's visibility timeout. Killed: dropping ReadError+RemoteProtocolError from the request-sent tuple (2 failures), retrying instead of breaking (4), treating 4xx as retryable (1), scalar timeout (2). tenant_account_v2/tests.py (+8) test_group_from_another_org_is_never_mailed was vacuous: its foreign group had no members, so it passed on an empty recipient list rather than on the org filter, and stayed green with that filter deleted. Its group now has a member who also belongs to the sharing org, which is the only arrangement that makes the filter load-bearing -- users belong to any number of orgs here. Added: a group member outside the org, an actor outside the org, a resource from another org, the org-wide revoke case, an owner inside a revoked group, the membership REMOVED direction, and recipient re-validation on membership. Note on what these pin. OrganizationGroup has no org-scoped default manager ("org filtering is explicit on every query"), so _groups_in_org's filter is genuinely load-bearing and is now pinned. OrganizationMember and Workflow do have one, so the explicit filters in _live_member_users, _get_user and _load_resource are defence-in-depth and their removal is behaviour- preserving -- these tests pin the outcome, not those lines. The docstrings say so rather than implying coverage they do not have. The one case _load_resource's filter genuinely guards is AgenticProject, whose manager deliberately spans orgs; that model is cloud-only and cannot be exercised here. Verified through the rig, not just pytest: tox -e groups -- unit-workers (1443 passed, 1 skipped) and integration-backend, where the new backend tests land because conftest auto-marks TestCase as integration. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/tenant_account_v2/tests.py | 141 ++++++++++ workers/tests/test_group_notification_post.py | 253 ++++++++++++++++++ 2 files changed, 394 insertions(+) create mode 100644 workers/tests/test_group_notification_post.py diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index eef404389c..ee4eeea45a 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -26,6 +26,7 @@ from workflow_manager.workflow_v2.models.workflow import Workflow from tenant_account_v2.group_notification_service import ( + ResourceNotFoundError, send_membership_changed, send_resource_shared, ) @@ -558,16 +559,156 @@ def test_revoke_mails_members_although_the_share_row_is_gone(self) -> None: self.assertEqual(kwargs["resource_name"], "wf-1") def test_group_from_another_org_is_never_mailed(self) -> None: + """The foreign group's member is deliberately also an org-A member. + + Users belong to any number of orgs here, so without that the group has + no resolvable recipients and the test passes on an empty list rather + than on the org filter -- green even with the filter deleted. + """ other_org = Organization.objects.create( name="org-b", display_name="Org B", organization_id="org-b" ) foreign_group = OrganizationGroup.objects.create( organization=other_org, name="Foreign", created_by=self.owner ) + dual = _make_user("dual@example.com") + OrganizationMember.objects.create(organization=self.org, user=dual, role="user") + OrganizationMember.objects.create(organization=other_org, user=dual, role="user") + GroupMembership.objects.create(group=foreign_group, user=dual) + for action in (ShareAction.SHARED.value, ShareAction.REVOKED.value): self._send(group_ids=[foreign_group.id], share_action=action) + self.assertEqual(self._mailed(), []) + + def test_group_member_outside_the_org_is_never_mailed(self) -> None: + """A group row that outlives the org membership must not produce mail. + + Pins the outcome, not the layer: ``OrganizationMember``'s default + manager is org-scoped, so deleting the explicit filter in + ``_live_member_users`` changes nothing. ``OrganizationGroup`` has no + such manager, which is why the group-level filter IS pinnable -- see + ``test_group_from_another_org_is_never_mailed``. + """ + other_org = Organization.objects.create( + name="org-d", display_name="Org D", organization_id="org-d" + ) + # A member of ANOTHER org, not of no org: a user with no membership row + # at all is excluded by the table rather than by the org clause, which + # would leave this test green with the filter deleted. + stranger = _make_user("stranger@example.com") + OrganizationMember.objects.create( + organization=other_org, user=stranger, role="user" + ) + GroupMembership.objects.create(group=self.group, user=stranger) + set_resource_share_groups(self.workflow, [self.group.id]) + self._send(group_ids=[self.group.id]) + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + + def test_actor_outside_the_org_is_not_resolved(self) -> None: + """The actor's name and email render into the outgoing mail, so an + actor from another org must not resolve. Outcome-level, like the + recipient case above: the org-scoped manager enforces it either way. + """ + other_org = Organization.objects.create( + name="org-e", display_name="Org E", organization_id="org-e" + ) + # Again a member of another org rather than of none, so the org clause + # is the only thing that can exclude them. + foreign_actor = _make_user("foreign-actor@example.com") + OrganizationMember.objects.create( + organization=other_org, user=foreign_actor, role="user" + ) + # The grant direction drops a group with no live share row, which would + # stop the mail before the actor is ever resolved. + set_resource_share_groups(self.workflow, [self.group.id]) + send_resource_shared( + organization=self.org, + group_ids=[self.group.id], + actor_id=foreign_actor.pk, + resource_kind="workflow", + resource_id=str(self.workflow.pk), + share_action=ShareAction.SHARED.value, + revoked_at=None, + ) + self.service.send_group_resource_shared_notification.assert_not_called() + + def test_resource_from_another_org_is_not_resolved(self) -> None: + """A resource id belonging to another org must not resolve. + + For ``Workflow`` the org-scoped manager already enforces this, so this + pins the outcome rather than the explicit filter. That filter exists + for ``AgenticProject``, whose manager deliberately spans orgs -- a + cloud-only model, so the case it guards cannot be exercised here. + """ + other_org = Organization.objects.create( + name="org-c", display_name="Org C", organization_id="org-c" + ) + foreign_wf = Workflow.objects.create( + workflow_name="wf-other", organization=other_org, created_by=self.owner + ) + with self.assertRaises(ResourceNotFoundError): + send_resource_shared( + organization=self.org, + group_ids=[self.group.id], + actor_id=self.owner.pk, + resource_kind="workflow", + resource_id=str(foreign_wf.pk), + share_action=ShareAction.SHARED.value, + revoked_at=None, + ) self.service.send_group_resource_shared_notification.assert_not_called() + def test_revoke_on_an_org_shared_resource_mails_nobody(self) -> None: + """Nobody lost access, so nobody is told. + + The short-circuit that skips hydrating every org member to reach this + answer is an optimisation, not a behaviour change -- removing it leaves + this assertion green. Only a query count would pin that half. + """ + self.workflow.shared_to_org = True + self.workflow.save(update_fields=["shared_to_org"]) + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + self.assertEqual(self._mailed(), []) + + def test_revoke_does_not_tell_an_owner_they_lost_access(self) -> None: + """Owners sit outside ``compute_effective_members``, so they have to be + added back explicitly or an owner inside a revoked group is mailed a + false removal notice. + """ + GroupMembership.objects.create(group=self.group, user=self.owner) + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + + def test_membership_removal_is_mailed_as_a_removal(self) -> None: + """The ADDED direction was the only one exercised, so hardcoding the + action passed every test while telling removed users they were added. + """ + send_membership_changed( + organization=self.org, + group_id=self.group.id, + actor_id=self.owner.pk, + membership_action=MembershipAction.REMOVED.value, + user_ids=[self.member.pk], + ) + kwargs = self.service.send_group_membership_notification.call_args.kwargs + self.assertEqual(kwargs["membership_action"], "removed") + + def test_membership_recipients_are_revalidated_against_the_org(self) -> None: + """Leaving a group does not remove someone from the org, and leaving the + org does not delete their group rows -- so the recipient list is filtered + on OrganizationMember rather than taken from the payload. + """ + stranger = _make_user("ex@example.com") + send_membership_changed( + organization=self.org, + group_id=self.group.id, + actor_id=self.owner.pk, + membership_action=MembershipAction.REMOVED.value, + user_ids=[self.member.pk, stranger.pk], + ) + kwargs = self.service.send_group_membership_notification.call_args.kwargs + self.assertEqual([u.email for u in kwargs["recipients"]], ["member@example.com"]) + def test_revoke_skips_members_who_joined_after_the_cutoff(self) -> None: revoked_at = timezone.now() latecomer = GroupMembership.objects.create(group=self.group, user=self.outsider) diff --git a/workers/tests/test_group_notification_post.py b/workers/tests/test_group_notification_post.py new file mode 100644 index 0000000000..d719f734c1 --- /dev/null +++ b/workers/tests/test_group_notification_post.py @@ -0,0 +1,253 @@ +"""The group-notification worker leg is entirely failure-path logic. + +``_post_group_notification`` is the only thing standing between a transient +backend blip and a permanently unsent email, and every branch in it encodes a +different judgement about whether re-posting is safe. Two of those judgements +are easy to get backwards: + +* A response the backend never sent (connect refused, DNS gone) is safe to + re-post -- nothing happened. +* A response lost *after* the request arrived is not. The backend does not stop + when the client disconnects, and the send path mails group by group with no + checkpoint, so a re-post re-mails everyone who already received it. + +These pin which exception lands on which side, plus the attempt cap, the +sub-500 break, and the fact that the timeout is per-phase rather than scalar -- +httpx applies a scalar timeout to connect, write and read separately, so a +scalar one silently triples the task's worst-case wall time and can push it +past the queue's visibility timeout. +""" + +from __future__ import annotations + +from unittest.mock import MagicMock, patch + +import httpx +import pytest +from notification.tasks import ( + _GROUP_NOTIFICATION_ATTEMPTS, + _GROUP_NOTIFICATION_TIMEOUT, + _post_group_notification, + notify_group_membership_changed, + notify_resource_shared_with_group, +) + +_ENV = { + "INTERNAL_API_BASE_URL": "http://backend/internal", + "INTERNAL_SERVICE_API_KEY": "k", +} + + +def _response(status: int) -> MagicMock: + return MagicMock(status_code=status, text="body") + + +class _Client: + """Stand-in for ``httpx.Client`` that records posts and replays outcomes.""" + + def __init__(self, outcomes): + self.outcomes = list(outcomes) + self.calls = [] + + def __call__(self, *a, **kw): + return self + + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def post(self, url, **kwargs): + self.calls.append((url, kwargs)) + outcome = self.outcomes.pop(0) if self.outcomes else _response(200) + if isinstance(outcome, Exception): + raise outcome + return outcome + + +def _run(outcomes, endpoint="resource-shared", payload=None): + """Drive one ``_post_group_notification`` with ``outcomes`` per attempt.""" + client = _Client(outcomes) + with ( + patch.dict("os.environ", _ENV, clear=False), + patch("notification.tasks.httpx.Client", client), + patch("notification.tasks.time.sleep") as sleep, + ): + raised = None + try: + _post_group_notification(endpoint, "org-a", payload or {"x": 1}) + except Exception as e: # noqa: BLE001 + raised = e + return client, raised, sleep + + +class TestRetryClassification: + def test_success_posts_once_and_returns(self): + client, raised, _ = _run([_response(200)]) + assert raised is None + assert len(client.calls) == 1 + + def test_server_error_exhausts_the_attempt_cap_then_raises(self): + client, raised, sleep = _run([_response(503)] * _GROUP_NOTIFICATION_ATTEMPTS) + assert isinstance(raised, RuntimeError) + assert len(client.calls) == _GROUP_NOTIFICATION_ATTEMPTS + assert sleep.call_count == _GROUP_NOTIFICATION_ATTEMPTS - 1 + + def test_client_error_is_terminal_after_one_post(self): + # A rejected payload will be rejected again; retrying only burns budget. + client, raised, sleep = _run([_response(400)]) + assert isinstance(raised, RuntimeError) + assert len(client.calls) == 1 + assert sleep.call_count == 0 + + @pytest.mark.parametrize( + "exc", + [ + httpx.ReadTimeout("read"), + httpx.WriteTimeout("write"), + httpx.ReadError("read err"), + httpx.RemoteProtocolError("server disconnected"), + ], + ) + def test_request_sent_outcome_unknown_is_never_re_posted(self, exc): + """Each of these means the backend already has the request. + + Re-posting would re-mail every group that already succeeded, so the + in-process attempts must end after one post. + """ + client, raised, sleep = _run([exc]) + assert isinstance(raised, RuntimeError) + assert len(client.calls) == 1, f"{type(exc).__name__} was re-posted" + assert sleep.call_count == 0 + + @pytest.mark.parametrize( + "exc", + [ + httpx.ConnectTimeout("connect"), + httpx.ConnectError("refused"), + httpx.PoolTimeout("pool"), + ], + ) + def test_request_never_left_is_retried(self, exc): + """Nothing reached the backend, so the full attempt cap is correct.""" + client, raised, _ = _run([exc] * _GROUP_NOTIFICATION_ATTEMPTS) + assert isinstance(raised, RuntimeError) + assert len(client.calls) == _GROUP_NOTIFICATION_ATTEMPTS + + def test_recovers_when_a_later_attempt_succeeds(self): + client, raised, _ = _run([_response(502), _response(200)]) + assert raised is None + assert len(client.calls) == 2 + + +class TestRequestShape: + def test_missing_credentials_raise_before_any_post(self): + client = _Client([]) + with ( + patch.dict("os.environ", {"INTERNAL_API_BASE_URL": "", "INTERNAL_SERVICE_API_KEY": ""}), + patch("notification.tasks.httpx.Client", client), + ): + with pytest.raises(RuntimeError): + _post_group_notification("resource-shared", "org-a", {}) + assert client.calls == [] + + def test_url_auth_and_tenant_header(self): + client, _, _ = _run([_response(200)], endpoint="membership-changed") + url, kwargs = client.calls[0] + assert url == "http://backend/internal/v1/group-notification/membership-changed/" + assert kwargs["headers"]["Authorization"] == "Bearer k" + # Without this the backend resolves no tenant and every org-scoped + # query comes back empty -- a silent no-op rather than an error. + assert kwargs["headers"]["X-Organization-ID"] == "org-a" + + def test_timeout_is_per_phase_not_scalar(self): + """A scalar timeout is applied to connect, write AND read separately. + + Passing one would let a single post spend it three times over, which is + what pushed this task past the consumer's visibility timeout. + """ + client, _, _ = _run([_response(200)]) + timeout = client.calls[0][1]["timeout"] + assert isinstance(timeout, httpx.Timeout) + assert timeout is _GROUP_NOTIFICATION_TIMEOUT + assert timeout.connect < timeout.read + + def test_task_wall_time_stays_under_the_visibility_timeout(self): + """The bound the chart and compose comments defer to. + + VT is 300s for this worker in both deployments; the heartbeat is frozen + for the task's duration, so health-stale (360s) is the other ceiling. + """ + t = _GROUP_NOTIFICATION_TIMEOUT + per_post = t.connect + t.write + t.read + t.pool + worst = per_post * _GROUP_NOTIFICATION_ATTEMPTS + assert worst < 300, f"worst case {worst}s exceeds the 300s visibility timeout" + + +class TestTaskPayloads: + def test_resource_shared_task_sends_every_field_the_endpoint_requires(self): + client = _Client([_response(200)]) + with ( + patch.dict("os.environ", _ENV, clear=False), + patch("notification.tasks.httpx.Client", client), + ): + notify_resource_shared_with_group( + group_ids=[2, 5], + actor_id=7, + resource_kind="workflow", + resource_id="wf-1", + organization_id="org-a", + share_action="revoked", + revoked_at="2026-01-01T00:00:00+00:00", + ) + url, kwargs = client.calls[0] + assert url.endswith("/resource-shared/") + assert kwargs["json"] == { + "group_ids": [2, 5], + "actor_id": 7, + "resource_kind": "workflow", + "resource_id": "wf-1", + "share_action": "revoked", + "revoked_at": "2026-01-01T00:00:00+00:00", + } + + def test_share_direction_sends_a_null_cutoff_rather_than_omitting_it(self): + # The endpoint requires the key on both directions; omitting it is a 400. + client = _Client([_response(200)]) + with ( + patch.dict("os.environ", _ENV, clear=False), + patch("notification.tasks.httpx.Client", client), + ): + notify_resource_shared_with_group( + group_ids=[2], + actor_id=7, + resource_kind="workflow", + resource_id="wf-1", + organization_id="org-a", + share_action="shared", + revoked_at=None, + ) + assert client.calls[0][1]["json"]["revoked_at"] is None + + def test_membership_task_payload(self): + client = _Client([_response(200)]) + with ( + patch.dict("os.environ", _ENV, clear=False), + patch("notification.tasks.httpx.Client", client), + ): + notify_group_membership_changed( + group_id=3, + actor_id=7, + membership_action="removed", + user_ids=[11, 12], + organization_id="org-a", + ) + url, kwargs = client.calls[0] + assert url.endswith("/membership-changed/") + assert kwargs["json"] == { + "group_id": 3, + "actor_id": 7, + "membership_action": "removed", + "user_ids": [11, 12], + } From 384aac2bb80a319fdf4276a748b44659ec7108c9 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 00:12:17 +0530 Subject: [PATCH 17/28] UN-3494 [FIX] Stop telling retained users they lost access A revoke mailed "your access was removed" to two sets of people who had lost nothing. An org admin reaches every resource because for_user hands admins the whole queryset, and a frictionless adapter is admitted unconditionally. The retained set was built from share rows alone, so neither route appeared in it. Both routes now sit in sharing_helpers beside compute_effective_members, so the two places that decide who lost access -- the group fan-out and the direct share view -- share one answer rather than each carrying its own partial copy. The direct share view had the identical hole. The admin check goes through AuthenticationController instead of comparing to a role string, which differs between the OSS and auth0 plugins. Each line was verified by breaking it: dropping the admin add-back fails the new admin test, and dropping the frictionless route fails 2 of the 6 parametrized retention cases. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/resource_share_views.py | 18 +++++-- .../group_notification_service.py | 40 ++++++++------ backend/tenant_account_v2/sharing_helpers.py | 30 +++++++++++ .../test_sharing_retention.py | 52 +++++++++++++++++++ backend/tenant_account_v2/tests.py | 16 ++++++ 5 files changed, 134 insertions(+), 22 deletions(-) create mode 100644 backend/tenant_account_v2/test_sharing_retention.py diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 660af7dd91..ebe1e21927 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -68,14 +68,22 @@ def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: """ if not users: return [] - if getattr(instance, "shared_to_org", False): - # Org-wide share still covers everyone — nobody lost access, and - # answering it via ``compute_effective_members`` would hydrate every - # member of the org to say so. + from tenant_account_v2.sharing_helpers import ( + access_survives_share_changes, + compute_effective_members, + org_admin_user_ids, + ) + + if access_survives_share_changes(instance): + # Org-wide or frictionless: access never depended on the share, so + # nobody lost anything — and answering it via + # ``compute_effective_members`` would hydrate the whole org to say so. return [] - from tenant_account_v2.sharing_helpers import compute_effective_members retained = {member["user_id"] for member in compute_effective_members(instance)} + # Admins are outside compute_effective_members but ``for_user`` hands them + # every resource in the org, so a revoke takes nothing from them. + retained |= org_admin_user_ids(getattr(instance, "organization", None)) return [user for user in users if user.pk not in retained] diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index bed52f9667..3803c78211 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -98,7 +98,7 @@ def send_resource_shared( shared.type, ) return - retained = _retained_user_ids(shared.instance, share_action) + retained = _retained_user_ids(organization, shared.instance, share_action) if retained is None: return for group in _groups_to_mail(organization, group_ids, shared.instance, share_action): @@ -193,35 +193,41 @@ def _get_user(organization: Organization, user_id: int) -> User | None: return member.user if member else None -def _retained_user_ids(resource: Any, share_action: str) -> set[int] | None: +def _retained_user_ids( + organization: Organization, resource: Any, share_action: str +) -> set[int] | None: """Users who still reach ``resource``; empty on the share direction. - A revoked group's members may keep access through another group, a direct - share or an org-wide share — telling them it was removed would be wrong, - and the revoke email also repoints their CTA at the dashboard. Owners sit - outside ``compute_effective_members`` by design, so add them back: an owner - in the revoked group has lost nothing. + A revoked group's members may keep access by a route the revoke did not + touch — another group, a direct share, ownership, or being an org admin, + who reaches every resource in the org. Telling any of them their access was + removed would be wrong, and the revoke email also repoints their CTA at the + dashboard. ``compute_effective_members`` covers the share routes only, so + owners and admins are added back explicitly. - ``None`` means an org-wide share still covers everyone, so the caller skips - the fan-out entirely. + ``None`` means access never depended on the share at all, so the caller + skips the fan-out entirely. """ if share_action != ShareAction.REVOKED.value: return set() - if getattr(resource, "shared_to_org", False): - # Org-wide share still covers everyone — nobody lost access, and - # answering it via ``compute_effective_members`` would hydrate every - # member of the org to say so (same guard as the direct-share path). + from tenant_account_v2.sharing_helpers import ( + access_survives_share_changes, + compute_effective_members, + org_admin_user_ids, + ) + + if access_survives_share_changes(resource): logger.info( - "group-notification: revoke on an org-shared resource %s — " - "nobody lost access, no mail", + "group-notification: revoke on %s, whose access does not depend on " + "shares — nobody lost access, no mail", resource.pk, ) return None - from tenant_account_v2.sharing_helpers import compute_effective_members - return {member["user_id"] for member in compute_effective_members(resource)} | { + retained = {member["user_id"] for member in compute_effective_members(resource)} | { owner.pk for owner in resource.owners() } + return retained | org_admin_user_ids(organization) def _groups_to_mail( diff --git a/backend/tenant_account_v2/sharing_helpers.py b/backend/tenant_account_v2/sharing_helpers.py index 36887b1331..3acbd8c163 100644 --- a/backend/tenant_account_v2/sharing_helpers.py +++ b/backend/tenant_account_v2/sharing_helpers.py @@ -261,6 +261,36 @@ def serialize_owner_refs(resource_obj: Any) -> list[dict[str, Any]]: return [{"id": user.pk, "email": user.email} for user in resource_obj.owners()] +def access_survives_share_changes(resource_obj: Any) -> bool: + """True when access to ``resource_obj`` does not depend on shares at all. + + Revoking a share removes nothing in that case, so nobody should be told it + did. ``shared_to_org`` covers every org member; ``is_friction_less`` is the + adapter equivalent -- ``AdapterInstance.for_user`` admits it unconditionally. + """ + return bool(getattr(resource_obj, "shared_to_org", False)) or bool( + getattr(resource_obj, "is_friction_less", False) + ) + + +def org_admin_user_ids(organization: Any) -> set[int]: + """Users who reach every resource in ``organization`` as admins. + + ``for_user`` returns the whole queryset for an org admin, so a revoke never + takes their access away and they must not be mailed about losing it. The + role STRING is plugin-dependent (it differs between the OSS and auth0 + plugins), so this goes through the auth controller rather than comparing to + a literal. + """ + from account_v2.authentication_controller import AuthenticationController + + controller = AuthenticationController() + rows = OrganizationMember.objects.filter(organization=organization).values_list( + "user_id", "role" + ) + return {uid for uid, role in rows if controller.is_admin_by_role(role)} + + def compute_effective_members(resource_obj: Any) -> list[dict[str, Any]]: """Compute effective members of a shareable resource. diff --git a/backend/tenant_account_v2/test_sharing_retention.py b/backend/tenant_account_v2/test_sharing_retention.py new file mode 100644 index 0000000000..1c051eeb9c --- /dev/null +++ b/backend/tenant_account_v2/test_sharing_retention.py @@ -0,0 +1,52 @@ +"""Which resources have access that does not depend on shares at all. + +Revoking a share from such a resource removes nothing, so nobody may be told it +did. Two routes qualify and they are easy to miss because neither is a share +row: ``shared_to_org`` admits every org member, and ``is_friction_less`` is the +adapter equivalent -- ``AdapterInstance.for_user`` admits it unconditionally, +alongside the share clauses. + +Pure predicate, so this runs in the rig's unit tier with no database. The +admin route is the third one and is user-level rather than resource-level, so +it lives in ``org_admin_user_ids`` and is covered by the DB tests. +""" + +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +from tenant_account_v2.sharing_helpers import access_survives_share_changes + + +class TestAccessSurvivesShareChanges: + @pytest.mark.parametrize( + "resource, expected, why", + [ + (SimpleNamespace(shared_to_org=True), True, "org-wide share"), + (SimpleNamespace(is_friction_less=True), True, "frictionless adapter"), + ( + SimpleNamespace(shared_to_org=False, is_friction_less=True), + True, + "frictionless without an org share still admits everyone", + ), + ( + SimpleNamespace(shared_to_org=True, is_friction_less=False), + True, + "org share without frictionless still admits everyone", + ), + ( + SimpleNamespace(shared_to_org=False, is_friction_less=False), + False, + "neither route: access really does depend on the share", + ), + ( + SimpleNamespace(), + False, + "a model carrying neither field, e.g. Workflow, is share-dependent", + ), + ], + ) + def test_routes(self, resource, expected, why): + assert access_survives_share_changes(resource) is expected, why diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index ee4eeea45a..77bc47aef0 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -679,6 +679,22 @@ def test_revoke_does_not_tell_an_owner_they_lost_access(self) -> None: self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + def test_revoke_does_not_tell_an_org_admin_they_lost_access(self) -> None: + """An admin reaches every resource in the org via ``for_user``, so a + revoke takes nothing from them. + + The admin ROLE STRING differs between the OSS and auth0 auth plugins, + so the predicate is patched rather than resolved for real. + """ + GroupMembership.objects.create(group=self.group, user=self.admin) + with patch( + "account_v2.authentication_controller.AuthenticationController" + ".is_admin_by_role", + side_effect=lambda role: role == "admin", + ): + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + self.assertEqual(self._mailed(), [("Team", ["member@example.com"])]) + def test_membership_removal_is_mailed_as_a_removal(self) -> None: """The ADDED direction was the only one exercised, so hardcoding the action passed every test while telling removed users they were added. From 665b0a936049a4b7d578dfedab63f7ca0ea34733 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 15:32:54 +0530 Subject: [PATCH 18/28] UN-3494 [FIX] Close the gaps two independent code reviews found Two /code-review passes over the pushed state found ten real issues here. Two more were real but not fixed -- flagged below with why. - LookupDefinition was missing from the group-shareable resource registry. Its ViewSet already exposes the group-share action, so a group share on a lookup resolved no notification type and silently logged a warning instead of mailing anyone. - share() diffed both sharing axes even when a request's payload touched only one. A concurrent request changing the untouched axis landed inside that window and got attributed to the wrong actor's name in the notification. Now each axis is only read, diffed and notified when this request's own payload actually names it. - The worker's "already reached the backend, don't retry" exception tuple had ReadError but not its write-side sibling WriteError -- a mid-write socket failure retried the full attempt cap instead of stopping after one, same as every other request-sent-but-outcome-unknown case already does. - A group-membership add validated "not already a member" once, then wrote with ignore_conflicts=True. A concurrent add for the same user landed silently as a no-op, and the notification still fired for the user this request never actually added. Re-checks membership immediately before the write to shrink that window. - The revoke-path retention logic (who still has access another way: a group, ownership, an org-wide share, admin) was hand-rolled twice, once per call site, with owners present in one copy and silently absent from the other. Consolidated into sharing_helpers.retained_user_ids, used by both. - The group fan-out queried OrganizationMember once per group being mailed. Batched into one query across the whole group list per notification event. - _post_group_notification and onApplyCoOwners were both well over the 30-line function cap; module docstrings across three files cited ticket numbers rather than staying purpose-only. Extracted the retry-attempt shape and the mutation phase into their own functions; dropped the ticket references. - Co-owner demotion never checked whether the demoted user still reaches the resource another way (a group, a direct share, org-wide, or being an org admin) before mailing "your access was removed" -- the same class of bug the group-revoke and direct-share paths already guard against, just never extended to this call site. This PR had only added a docstring here; the bug predates it. Fixed by reusing the same retained_user_ids check, with a mutation-verified test pinning it. Flagged, not fixed -- real, disproportionate to fix here: - _post_group_notification hand-rolls a retry loop that resembles workers/shared/clients/base_client.py's BaseAPIClient. Different library (requests/urllib3, not httpx), no per-phase timeout, no request-sent classification -- the two things this PR already tuned. Swapping would risk regressing both for a cosmetic duplication. - group_notification_service.py's resource-type mapping duplicates what each ViewSet's own get_notification_resource_type computes. A proper fix reaches into five files this PR never touched to refactor an already-shipped feature; out of scope for a remediation pass. One reported finding did not hold up: a claim that a failed roster refresh leaves the co-owner modal showing stale data on reopen. Reopening always refetches fresh (handleCoOwner), so the claimed path does not exist. Verified: tox unit-backend (1294), unit-workers (1445), integration-backend (583) all pass; the sendgrid-plugin collection error is the known, expected local-only gap. New test mutation-verified. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/membership_views.py | 9 +- backend/permissions/resource_share_views.py | 46 ++++--- .../tests/test_owner_management.py | 46 ++++--- .../group_notification_service.py | 83 +++++++----- backend/tenant_account_v2/group_views.py | 28 ++-- backend/tenant_account_v2/internal_views.py | 2 +- .../tenant_account_v2/share_notifications.py | 7 +- .../tenant_account_v2/shareable_resources.py | 8 ++ backend/tenant_account_v2/sharing_helpers.py | 24 ++++ frontend/src/hooks/useCoOwnerManagement.jsx | 86 +++++++------ workers/notification/tasks.py | 120 ++++++++++-------- workers/tests/test_group_notification_post.py | 6 +- 12 files changed, 294 insertions(+), 171 deletions(-) diff --git a/backend/permissions/membership_views.py b/backend/permissions/membership_views.py index 329c6a3aa5..b1b9694db8 100644 --- a/backend/permissions/membership_views.py +++ b/backend/permissions/membership_views.py @@ -8,7 +8,7 @@ from rest_framework.decorators import action from rest_framework.request import Request from rest_framework.response import Response -from tenant_account_v2.sharing_helpers import serialize_owner_refs +from tenant_account_v2.sharing_helpers import retained_user_ids, serialize_owner_refs from permissions.membership_serializers import AddOwnerSerializer, RemoveOwnerSerializer @@ -120,6 +120,13 @@ def _notify_owner_removed(self, resource: Any, user: User, actor: Any) -> None: ctx = self._notification_context(resource) if ctx is None: return + # The OWNER row is already gone by the time this runs (serializer.save() + # ran first) -- but the demoted user may still reach the resource another + # way (a group, a direct share, an org-wide share, org admin), same as + # the direct-share and group-revoke paths already check. + retained = retained_user_ids(resource) + if retained is not None and user.pk in retained: + return resource_type, resource_name = ctx try: notification_plugin["service_class"]().send_access_removed_notification( diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index ebe1e21927..8fe0e86980 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -68,22 +68,13 @@ def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: """ if not users: return [] - from tenant_account_v2.sharing_helpers import ( - access_survives_share_changes, - compute_effective_members, - org_admin_user_ids, - ) + from tenant_account_v2.sharing_helpers import retained_user_ids - if access_survives_share_changes(instance): + retained = retained_user_ids(instance) + if retained is None: # Org-wide or frictionless: access never depended on the share, so - # nobody lost anything — and answering it via - # ``compute_effective_members`` would hydrate the whole org to say so. + # nobody lost anything. return [] - - retained = {member["user_id"] for member in compute_effective_members(instance)} - # Admins are outside compute_effective_members but ``for_user`` hands them - # every resource in the org, so a revoke takes nothing from them. - retained |= org_admin_user_ids(getattr(instance, "organization", None)) return [user for user in users if user.pk not in retained] @@ -144,8 +135,21 @@ def share(self, request: Request, pk: str | None = None) -> Response: resource = self.get_object() # type: ignore[attr-defined] desired = _extract_desired_share_state(request.data) - users_before = self._read_axis(resource, "shared_users") - groups_before = self._read_axis(resource, "shared_groups") + # Only read an axis this request actually touches. authorize_and_commit + # takes long enough (auth checks, a DB write) that a concurrent request + # changing an axis this one left alone would otherwise land inside the + # window and get diffed as if this request made the change — the wrong + # actor's name in the notification. + users_before = ( + self._read_axis(resource, "shared_users") + if "shared_users" in desired + else set() + ) + groups_before = ( + self._read_axis(resource, "shared_groups") + if "shared_groups" in desired + else set() + ) ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) @@ -156,8 +160,16 @@ def share(self, request: Request, pk: str | None = None) -> Response: # Only the two per-recipient axes notify. ``shared_to_org`` is left out # deliberately: a toggle has no recipient list short of the whole org, # and it is read below as a reason someone KEPT access, not lost it. - users_after = self._read_axis(resource, "shared_users") - groups_after = self._read_axis(resource, "shared_groups") + users_after = ( + self._read_axis(resource, "shared_users") + if "shared_users" in desired + else users_before + ) + groups_after = ( + self._read_axis(resource, "shared_groups") + if "shared_groups" in desired + else groups_before + ) notify_resource_group_share_changed( resource=resource, added=groups_after - groups_before, diff --git a/backend/permissions/tests/test_owner_management.py b/backend/permissions/tests/test_owner_management.py index 85a36ad8fb..af70b469e7 100644 --- a/backend/permissions/tests/test_owner_management.py +++ b/backend/permissions/tests/test_owner_management.py @@ -16,7 +16,6 @@ import pytest from account_v2.models import User from django.test import TestCase -from permissions.roles import ResourceRole from rest_framework import status from rest_framework.response import Response from rest_framework.test import APIRequestFactory, force_authenticate @@ -26,6 +25,7 @@ from permissions.membership_serializers import AddOwnerSerializer from permissions.permission import IsParentToolOwner +from permissions.roles import ResourceRole from permissions.tests.base import ( RESOURCE_SPECS, CoOwnerOrgTestMixin, @@ -124,7 +124,8 @@ def test_can_remove_when_multiple_owners(self) -> None: def test_remove_rejects_service_account(self) -> None: """Symmetric with the add-side guard: a service-account owner cannot be - removed, so it can't be stranded off the resource (UN-2202 review #7).""" + removed, so it can't be stranded off the resource (UN-2202 review #7). + """ svc = make_user("svc@example.com", is_service_account=True) self.workflow.memberships.create(user=svc, role=ResourceRole.OWNER) response = self._remove(self.owner, svc.pk) @@ -134,7 +135,8 @@ def test_remove_rejects_service_account(self) -> None: def test_shared_viewer_can_list_shared_users(self) -> None: """``list_of_shared_users`` is viewer-tier by design (spec §10): a shared user may open the owner popup. Guards against a re-tighten to IsOwner - (UN-2202 review #6).""" + (UN-2202 review #6). + """ self.workflow.memberships.create(user=self.viewer, role=ResourceRole.VIEWER) view = WorkflowViewSet.as_view({"get": "list_of_shared_users"}) request = self.factory.get("/x/") @@ -191,7 +193,8 @@ def test_has_members_mixin_accessors(self) -> None: def test_service_account_excluded_from_owner_roster(self) -> None: """A service-account OWNER row is a real grant but must not surface as a - removable co-owner or inflate the count badge (UN-2202 review #7).""" + removable co-owner or inflate the count badge (UN-2202 review #7). + """ svc = make_user("svc@example.com", is_service_account=True) self.workflow.memberships.create(user=self.coowner, role=ResourceRole.OWNER) self.workflow.memberships.create(user=svc, role=ResourceRole.OWNER) @@ -201,9 +204,7 @@ def test_service_account_excluded_from_owner_roster(self) -> None: self.assertEqual(self.workflow.co_owners_count(), 2) # the filter is display-only — the SA still genuinely owns the resource self.assertTrue( - self.workflow.memberships.filter( - user=svc, role=ResourceRole.OWNER - ).exists() + self.workflow.memberships.filter(user=svc, role=ResourceRole.OWNER).exists() ) def test_membership_save_derives_organization(self) -> None: @@ -216,7 +217,8 @@ def test_membership_save_derives_organization(self) -> None: class CrossResourceOwnerManagementTests(CoOwnerOrgTestMixin, TestCase): """The shared owner-management surface behaves identically for every OSS - shareable resource (AgenticProject is covered cloud-side).""" + shareable resource (AgenticProject is covered cloud-side). + """ def setUp(self) -> None: self._seed_org() @@ -253,7 +255,8 @@ class OwnerNotificationWiringTests(CoOwnerOrgTestMixin, TestCase): notifications with the right payload, and swallow notification failures so the owners request is never broken (best-effort). The resource-type hook is patched to a fixed value so the test does not depend on the cloud-only - notification plugin's conditional ``ResourceType`` import.""" + notification plugin's conditional ``ResourceType`` import. + """ def setUp(self) -> None: self._seed_org() @@ -311,6 +314,20 @@ def test_remove_fires_removed_notification_with_payload(self) -> None: self.assertEqual(kwargs["resource_id"], str(self.workflow.pk)) self.assertEqual(kwargs["resource_instance"], self.workflow) + def test_remove_skips_notification_when_demoted_owner_retains_access(self) -> None: + """Demoted from OWNER, but an org admin still reaches every resource -- + nothing was actually taken from them, so no "access removed" email. + """ + self.workflow.memberships.create(user=self.admin, role=ResourceRole.OWNER) + with patch( + "account_v2.authentication_controller.AuthenticationController" + ".is_admin_by_role", + side_effect=lambda role: role == "admin", + ): + response = self._remove(self.owner, self.admin.pk) + self.assertEqual(response.status_code, status.HTTP_204_NO_CONTENT) + self.service.send_access_removed_notification.assert_not_called() + def test_notification_failure_does_not_break_add(self) -> None: self.service.send_co_owner_added_notification.side_effect = RuntimeError("boom") response = self._add(self.owner, self.coowner.pk) @@ -399,9 +416,7 @@ def test_owner_exempt_from_default_adapter_clear(self) -> None: before = {self.owner.pk, self.coowner.pk, self.viewer.pk} after = {self.owner.pk} with ( - patch.object( - AdapterInstanceViewSet, "get_object", return_value=self.adapter - ), + patch.object(AdapterInstanceViewSet, "get_object", return_value=self.adapter), patch.object( AdapterInstanceViewSet, "_effective_member_ids", @@ -536,9 +551,7 @@ def test_adapter_create_grants_creator_ownership(self) -> None: from adapter_processor_v2.views import AdapterInstanceViewSet # Only the SDK context-window lookup (a provider-shaped call) is mocked. - with patch.object( - AdapterInstance, "get_context_window_size", return_value=4096 - ): + with patch.object(AdapterInstance, "get_context_window_size", return_value=4096): response = self._create( AdapterInstanceViewSet, { @@ -615,7 +628,8 @@ def test_api_deployment_create_grants_creator_ownership(self) -> None: def test_import_path_grants_creator_ownership(self) -> None: """The 7th grant site — ``create_tool_from_import_data`` — is a helper - the viewset sweep can't reach; pin it directly.""" + the viewset sweep can't reach; pin it directly. + """ from prompt_studio.prompt_studio_core_v2.models import CustomTool from prompt_studio.prompt_studio_core_v2.prompt_studio_helper import ( PromptStudioHelper, diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 3803c78211..831ca696bb 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -1,4 +1,4 @@ -"""Send-side logic for group-sharing email notifications (UN-3494 / mfbt UNS-848). +"""Send-side logic for group-sharing email notifications. Reached over the internal API by the notification worker. The enqueue side (:mod:`tenant_account_v2.share_notifications`) only records *what happened*; @@ -12,6 +12,7 @@ from __future__ import annotations import logging +from collections import defaultdict from dataclasses import dataclass from typing import TYPE_CHECKING, Any @@ -21,7 +22,11 @@ from django.db.models import QuerySet from plugins import get_plugin -from tenant_account_v2.models import OrganizationGroup, OrganizationMember +from tenant_account_v2.models import ( + GroupMembership, + OrganizationGroup, + OrganizationMember, +) from tenant_account_v2.share_notifications import MembershipAction, ShareAction from tenant_account_v2.shareable_resources import ShareableResource, descriptor_for_kind @@ -42,6 +47,7 @@ "connector_instance": "connector", "custom_tool": "text_extractor", "agentic_project": "agentic_project", + "lookup_definition": "lookup", } _ADAPTER_RESOURCE_TYPES = { "LLM": "llm", @@ -101,8 +107,12 @@ def send_resource_shared( retained = _retained_user_ids(organization, shared.instance, share_action) if retained is None: return - for group in _groups_to_mail(organization, group_ids, shared.instance, share_action): - recipients = _group_recipients(organization, group, retained, revoked_at) + groups = list(_groups_to_mail(organization, group_ids, shared.instance, share_action)) + recipients_by_group = _group_recipients_batch( + organization, groups, retained, revoked_at + ) + for group in groups: + recipients = recipients_by_group.get(group.pk, []) logger.info( "group-notification: task=notify_resource_shared_with_group " "group_id=%s action=%s recipient_count=%d", @@ -210,24 +220,16 @@ def _retained_user_ids( """ if share_action != ShareAction.REVOKED.value: return set() - from tenant_account_v2.sharing_helpers import ( - access_survives_share_changes, - compute_effective_members, - org_admin_user_ids, - ) + from tenant_account_v2.sharing_helpers import retained_user_ids - if access_survives_share_changes(resource): + retained = retained_user_ids(resource, organization) + if retained is None: logger.info( "group-notification: revoke on %s, whose access does not depend on " "shares — nobody lost access, no mail", resource.pk, ) - return None - - retained = {member["user_id"] for member in compute_effective_members(resource)} | { - owner.pk for owner in resource.owners() - } - return retained | org_admin_user_ids(organization) + return retained def _groups_to_mail( @@ -253,33 +255,52 @@ def _groups_to_mail( to_mail = [group for group in groups if group.pk in live] if len(to_mail) != len(groups): logger.info( - "group-notification: dropped %d of %d groups " - "(access revoked since enqueue)", + "group-notification: dropped %d of %d groups (access revoked since enqueue)", len(groups) - len(to_mail), len(groups), ) return to_mail -def _group_recipients( +def _group_recipients_batch( organization: Organization, - group: OrganizationGroup, + groups: list[OrganizationGroup], retained: set[int], joined_before: datetime | None = None, -) -> list[User]: - """Live members of ``group`` who did not keep access via ``retained``. - - ``joined_before`` (a revoke's timestamp) drops anyone who joined after the - access was taken away: they never held it through this group, so a - revocation notice would be about access they never had. +) -> dict[int, list[User]]: + """Live members of each of ``groups`` who did not keep access via ``retained``. + + One query across every group in the fan-out rather than one per group -- + a resource shared with N groups issued N ``OrganizationMember`` queries + before this, since ``joined_before`` (a revoke's timestamp) is the same + cutoff for every group being mailed in one call, so the membership lookup + batches cleanly. + + ``joined_before`` drops anyone who joined after the access was taken away: + they never held it through this group, so a revocation notice would be + about access they never had. """ - memberships = group.memberships + if not groups: + return {} + memberships = GroupMembership.objects.filter(group__in=groups) if joined_before is not None: memberships = memberships.filter(created_at__lte=joined_before) - users = _live_member_users( - organization, memberships.values_list("user_id", flat=True) - ) - return [user for user in users if user.pk not in retained] + user_ids_by_group: dict[int, set[int]] = defaultdict(set) + all_user_ids: set[int] = set() + for group_id, user_id in memberships.values_list("group_id", "user_id"): + user_ids_by_group[group_id].add(user_id) + all_user_ids.add(user_id) + users_by_id = { + user.pk: user for user in _live_member_users(organization, all_user_ids) + } + return { + group.pk: [ + users_by_id[uid] + for uid in user_ids_by_group.get(group.pk, ()) + if uid in users_by_id and uid not in retained + ] + for group in groups + } def _mail_group( diff --git a/backend/tenant_account_v2/group_views.py b/backend/tenant_account_v2/group_views.py index d5013b81b8..e8e7091aa0 100644 --- a/backend/tenant_account_v2/group_views.py +++ b/backend/tenant_account_v2/group_views.py @@ -155,18 +155,28 @@ def members(self, request: Request, pk: str | None = None) -> Response: serializer = GroupMemberAddSerializer(data=request.data, context={"group": group}) serializer.is_valid(raise_exception=True) user_ids_to_add: list[int] = serializer.validated_data["user_ids_to_add"] + # The serializer's "already a member" check ran at validation time, not + # at this write, so a concurrent request adding the same user in + # between would make our own insert a silent no-op (ignore_conflicts) + # while we still believe we added them. Re-check right before the + # write to keep that window as small as it can be. + already_members = set( + group.memberships.filter(user_id__in=user_ids_to_add).values_list( + "user_id", flat=True + ) + ) + newly_added_ids = [uid for uid in user_ids_to_add if uid not in already_members] GroupMembership.objects.bulk_create( - [GroupMembership(group=group, user_id=uid) for uid in user_ids_to_add], + [GroupMembership(group=group, user_id=uid) for uid in newly_added_ids], ignore_conflicts=True, ) - # The serializer already subtracts existing members, so nobody gets a - # second "you've been added" mail for a group they were already in. - notify_group_membership_changed( - group=group, - action=MembershipAction.ADDED, - user_ids=user_ids_to_add, - actor=request.user, - ) + if newly_added_ids: + notify_group_membership_changed( + group=group, + action=MembershipAction.ADDED, + user_ids=newly_added_ids, + actor=request.user, + ) return Response( {"added_user_ids": user_ids_to_add}, status=status.HTTP_201_CREATED, diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index d2cef14312..08c752041e 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -1,4 +1,4 @@ -"""Internal API views for group-sharing email notifications (UN-3494 / UNS-848). +"""Internal API views for group-sharing email notifications. Mounted under ``/internal/`` and gated by ``InternalAPIAuthMiddleware``. The notification worker calls these because ``workers/`` has no Django and every diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index c4565695e2..869dc7cab1 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -1,4 +1,4 @@ -"""Enqueue hooks for group-sharing email notifications (UN-3494 / mfbt UNS-848). +"""Enqueue hooks for group-sharing email notifications. Two events earn a group's members an email: a resource shared with or revoked from the group, and a user added to or removed from it. Both are dispatched @@ -6,9 +6,8 @@ The sending itself runs in ``workers/``, which is Django-free, so the worker task is a thin HTTP shim back to :mod:`tenant_account_v2.internal_views`; the -backend does the ORM and plugin work. Transport is the PG queue, the only one -there is (UN-4046) — the ``notifications`` queue the notification consumer -polls. +backend does the ORM and plugin work. Transport is the PG queue, the only +one there is — the ``notifications`` queue the notification consumer polls. A missing org or any dispatch error means no notification, never a broken share. Like the direct-user mail in ``ResourceShareManagementMixin`` and the diff --git a/backend/tenant_account_v2/shareable_resources.py b/backend/tenant_account_v2/shareable_resources.py index 2d55e0de0d..f0e7881780 100644 --- a/backend/tenant_account_v2/shareable_resources.py +++ b/backend/tenant_account_v2/shareable_resources.py @@ -49,6 +49,14 @@ class ShareableResource: "name", "id", ), + # ``lookups`` is cloud-only, same as ``agentic_studio_v1`` above. + ShareableResource( + "lookups", + "LookupDefinition", + "lookup_definition", + "name", + "id", + ), ) diff --git a/backend/tenant_account_v2/sharing_helpers.py b/backend/tenant_account_v2/sharing_helpers.py index 3acbd8c163..d396855431 100644 --- a/backend/tenant_account_v2/sharing_helpers.py +++ b/backend/tenant_account_v2/sharing_helpers.py @@ -343,6 +343,30 @@ def compute_effective_members(resource_obj: Any) -> list[dict[str, Any]]: return list(seen.values()) +def retained_user_ids(resource_obj: Any, organization: Any = None) -> set[int] | None: + """Every user id who still reaches ``resource_obj`` after a revoke. + + ``None`` means access never depended on shares at all (see + ``access_survives_share_changes``) -- the caller's cue that nobody lost + anything and no notification is owed. Shared by the direct-share and + group-share revoke paths so they can't independently drift on what counts + as "still has access". + """ + if access_survives_share_changes(resource_obj): + return None + retained = {member["user_id"] for member in compute_effective_members(resource_obj)} + # compute_effective_members deliberately excludes owners (they hold the + # resource, they aren't "shared with" it); a revoke never touches them. + retained |= {owner.pk for owner in resource_obj.owners()} + org = ( + organization + if organization is not None + else getattr(resource_obj, "organization", None) + ) + retained |= org_admin_user_ids(org) + return retained + + def _add_org_members(seen: dict[int, dict[str, Any]], resource_obj: Any) -> None: """Add org-wide members to ``seen`` (skips users already recorded).""" if not getattr(resource_obj, "shared_to_org", False): diff --git a/frontend/src/hooks/useCoOwnerManagement.jsx b/frontend/src/hooks/useCoOwnerManagement.jsx index 4a1b0978e5..def7ec4d5e 100644 --- a/frontend/src/hooks/useCoOwnerManagement.jsx +++ b/frontend/src/hooks/useCoOwnerManagement.jsx @@ -43,6 +43,42 @@ function buildApplyAlert( }; } +/** + * Run one Apply's add/remove calls. Attempts every user independently -- + * one rejection must not drop the rest or leave the modal contradicting the + * server. + */ +async function applyCoOwnerMutations( + service, + resourceId, + addUsers, + removeUsers, +) { + const failed = []; + let lastError = null; + const run = async (users, call) => { + for (const user of users) { + try { + await call(user.id); + } catch (err) { + failed.push(user); + lastError = err; + } + } + }; + // Adds first: the backend rejects removing the last owner, so a one-shot + // owner swap has to grow the roster before it shrinks it. + await run(addUsers, (id) => service.addCoOwner(resourceId, id)); + if (failed.length) { + // The roster never grew, so removing now can strip the very owner the + // swap was meant to replace. Report them rather than attempt them. + failed.push(...removeUsers); + } else { + await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); + } + return { failed, lastError }; +} + function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { const handleException = useExceptionHandler(); @@ -142,52 +178,24 @@ function useCoOwnerManagement({ service, setAlertDetails, onListRefresh }) { const onApplyCoOwners = useCallback( async (resourceId, { addUsers = [], removeUsers = [] }) => { const requestId = latestRequestRef.current; - // Attempt every user independently — one rejection must not drop the rest - // or leave the modal contradicting the server. - const failed = []; - let lastError = null; - const run = async (users, call) => { - for (const user of users) { - try { - await call(user.id); - } catch (err) { - failed.push(user); - lastError = err; - } - } - }; - // Adds first: the backend rejects removing the last owner, so a one-shot - // owner swap has to grow the roster before it shrinks it. - await run(addUsers, (id) => service.addCoOwner(resourceId, id)); - if (failed.length) { - // The roster never grew, so removing now can strip the very owner the - // swap was meant to replace. Report them rather than attempt them. - failed.push(...removeUsers); - } else { - await run(removeUsers, (id) => service.removeCoOwner(resourceId, id)); - } - // Reconverge the modal on true server state regardless of partial outcome. + const { failed, lastError } = await applyCoOwnerMutations( + service, + resourceId, + addUsers, + removeUsers, + ); + // Reconverge on true server state regardless of partial outcome. const outcome = await refreshCoOwnerData(resourceId, requestId); if (outcome !== "gone") { - // The mutations landed whatever the refresh said, and the list is - // page-scoped rather than modal-scoped. ("gone" already refreshed it.) - onListRefresh?.(); + onListRefresh?.(); // page-scoped list; "gone" already refreshed it } if (outcome === "stale") { - // The user has opened another resource, so closing this modal or - // alerting would hit that one instead. - return false; + return false; // another resource is open now, not our modal to alert } if (outcome === "gone") { - // The refresh closed the modal and raised its own alert; a summary on - // top of it would only mislead. - return true; + return true; // refresh already closed the modal and alerted } - // "ok" and "error" both fall through: the mutations landed either way and - // their outcomes are known without the refresh, so a clean apply closes - // the modal rather than stranding it. A partial failure still leaves the - // staged diff computed against an unrefreshed roster -- clearing that is - // not done here. + // "ok" and "error" both fall through: the mutations landed either way. setAlertDetails( buildApplyAlert( addUsers, diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index f521850f15..0cb826baa2 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -521,22 +521,10 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An _GROUP_NOTIFICATION_TIMEOUT = httpx.Timeout(connect=5.0, write=10.0, read=30.0, pool=5.0) -def _post_group_notification(endpoint: str, organization_id: str, payload: dict) -> None: - """POST a group-notification job to the backend and insist it succeeded. - - Unlike ``_mark_buffer_outcome`` this deliberately **raises** on failure: - nothing tracks an unsent group email, so a swallowed error would be a - silent drop. The raise leaves the message on the queue for redelivery, - bounded by the consumer's attempt cap. - - A sub-500 response ends the in-process attempts — a rejected payload will - be rejected again. Delivery is at-least-once: a response lost after the - backend already sent re-posts the *whole* payload, and the backend mails - group by group with no checkpoint, so a failure partway through the fan-out - re-mails the groups that already succeeded. The send path writes nothing, so - duplicate email is the only effect — but the bound is these 3 attempts times - the consumer's attempt cap, not one. - """ +def _build_group_notification_request( + endpoint: str, organization_id: str +) -> tuple[str, dict[str, str]]: + """URL and headers for one group-notification POST.""" base_url = os.getenv("INTERNAL_API_BASE_URL") api_key = os.getenv("INTERNAL_SERVICE_API_KEY") if not base_url or not api_key: @@ -551,38 +539,72 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) # org-scoped query comes back empty. "X-Organization-ID": organization_id, } + return url, headers + + +def _post_group_notification_once( + url: str, headers: dict[str, str], payload: dict +) -> tuple[bool, bool, str]: + """One POST attempt. Returns ``(succeeded, retryable, error)``. + + A response lost after the backend already received it is treated like a + sub-500 (not retryable): the backend does not stop when we disconnect and + mails group by group with no checkpoint, so re-posting would re-mail every + group that already succeeded. + """ + try: + with httpx.Client() as client: + response = client.post( + url, headers=headers, json=payload, timeout=_GROUP_NOTIFICATION_TIMEOUT + ) + except ( + httpx.ReadTimeout, + httpx.WriteTimeout, + httpx.ReadError, + httpx.WriteError, + httpx.RemoteProtocolError, + ) as e: + return False, False, f"timeout_after_send={e!r}" + except Exception as e: # noqa: BLE001 + return False, True, f"exception={e!r}" + if response.status_code == 200: + return True, False, "" + error = f"http_{response.status_code} body={response.text[:200]}" + return False, response.status_code >= 500, error + + +def _fail_group_notification(endpoint: str, organization_id: str, error: str) -> None: + """Log and raise once the attempt budget is exhausted. + + A duplicate email from the resulting redelivery is the only effect, since + the send path writes nothing. + """ + logger.error( + "metric=group_notification_post_failed_total endpoint=%s org_id=%s error=%s", + endpoint, + organization_id, + error, + ) + raise RuntimeError(f"Group notification {endpoint} failed: {error}") + + +def _post_group_notification(endpoint: str, organization_id: str, payload: dict) -> None: + """POST a group-notification job to the backend and insist it succeeded. + + Deliberately raises on failure -- nothing tracks an unsent group email, so + a swallowed error would be a silent drop. The raise leaves the message on + the queue for redelivery, bounded by the consumer's attempt cap. + """ + url, headers = _build_group_notification_request(endpoint, organization_id) last_error = "" for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): - try: - with httpx.Client() as client: - response = client.post( - url, - headers=headers, - json=payload, - timeout=_GROUP_NOTIFICATION_TIMEOUT, - ) - except ( - httpx.ReadTimeout, - httpx.WriteTimeout, - httpx.ReadError, - httpx.RemoteProtocolError, - ) as e: - # The request reached the backend, and the backend does not stop - # when we disconnect. Its outcome is unknown, and the send path has - # no checkpoint, so re-posting re-mails every group that already - # succeeded. End the in-process attempts like a sub-500 does. - # (The queue can still redeliver; only an idempotency key on the - # handler would bound that, which this PR does not add.) - last_error = f"timeout_after_send={e!r}" + succeeded, retryable, last_error = _post_group_notification_once( + url, headers, payload + ) + if succeeded: + return + if not retryable: break - except Exception as e: # noqa: BLE001 - last_error = f"exception={e!r}" - else: - if response.status_code == 200: - return - last_error = f"http_{response.status_code} body={response.text[:200]}" - if response.status_code < 500: - break if attempt < _GROUP_NOTIFICATION_ATTEMPTS: logger.warning( "Group notification %s attempt %d/%d failed (%s); retrying", @@ -592,13 +614,7 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) last_error, ) time.sleep(_GROUP_NOTIFICATION_RETRY_DELAY) - logger.error( - "metric=group_notification_post_failed_total endpoint=%s org_id=%s error=%s", - endpoint, - organization_id, - last_error, - ) - raise RuntimeError(f"Group notification {endpoint} failed: {last_error}") + _fail_group_notification(endpoint, organization_id, last_error) @worker_task(name="notify_resource_shared_with_group") diff --git a/workers/tests/test_group_notification_post.py b/workers/tests/test_group_notification_post.py index d719f734c1..7f99f1d1cb 100644 --- a/workers/tests/test_group_notification_post.py +++ b/workers/tests/test_group_notification_post.py @@ -107,6 +107,7 @@ def test_client_error_is_terminal_after_one_post(self): httpx.ReadTimeout("read"), httpx.WriteTimeout("write"), httpx.ReadError("read err"), + httpx.WriteError("write err"), httpx.RemoteProtocolError("server disconnected"), ], ) @@ -145,7 +146,10 @@ class TestRequestShape: def test_missing_credentials_raise_before_any_post(self): client = _Client([]) with ( - patch.dict("os.environ", {"INTERNAL_API_BASE_URL": "", "INTERNAL_SERVICE_API_KEY": ""}), + patch.dict( + "os.environ", + {"INTERNAL_API_BASE_URL": "", "INTERNAL_SERVICE_API_KEY": ""}, + ), patch("notification.tasks.httpx.Client", client), ): with pytest.raises(RuntimeError): From cc5782685560cd5cef9e77e52f986afe2e87b299 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 15:56:43 +0530 Subject: [PATCH 19/28] UN-3494 [MISC] Keep docker-compose.yaml out of this PR The only change left in it was a comment on worker-pg-notification. Reverted to main so this PR touches no compose file. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01LfrKHgSfDQxaXwJWrMxUWG --- docker/docker-compose.yaml | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index 845f646ce0..e89ca0e19d 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -589,9 +589,7 @@ services: - WORKER_PG_QUEUE_CONSUMER_CONCURRENCY=${PG_NOTIFICATION_CONCURRENCY:-2} # Log publisher — see the note on the backend service (UN-4046). - LOG_TRANSPORT=redis - # Several outbound HTTP POSTs per task -- the group-notification retry - # budget is derived at ``_GROUP_NOTIFICATION_ATTEMPTS`` in - # workers/notification/tasks.py. VT sits above a slow subscriber so a + # One outbound HTTP POST per task. VT sits above a slow subscriber so a # sibling replica cannot re-claim mid-POST and double-deliver; health-stale # above that. Mirrors the chart's workerPgNotification bounds. - WORKER_PG_QUEUE_CONSUMER_VT_SECONDS=${PG_NOTIFICATION_VT_SECONDS:-300} From 8cb9a5aa64eeb15c88f5c208d7102369600d4f69 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 17:42:52 +0530 Subject: [PATCH 20/28] UN-3494 [FIX] Show the real saved values when editing a pipeline notification Reopening a notification to edit it always showed a blank name/URL, BEARER auth, and an unchecked "notify on failures only" -- the actual saved values, regardless of what they were. formDetails started at DEFAULT_FORM_DETAILS on the component's first render; the real row only arrived one render later, via an effect. The antd-compatible Form shim seeds its fields from initialValues once, on its own first mount -- matching real antd -- which happens on that same first render, before the effect runs. The form always mounted on the blanks. A second effect tried to patch this with form.resetFields(), but the shim's resetFields() resets to an empty object rather than back to initialValues, so it could only re-blank the form, never repair it. This component remounts fresh every time Edit opens (NotificationModal renders DisplayNotifications in between), so editDetails is already the row being edited by the first render. Seeding formDetails from it via a lazy useState initializer fixes the timing directly and makes both effects unnecessary. Verified live: editing a notification now shows its real name, URL, authorization type, and notify-on-failures setting. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- .../notification-modal/CreateNotification.jsx | 28 ++++++++----------- 1 file changed, 11 insertions(+), 17 deletions(-) diff --git a/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx b/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx index 5ed4382b1a..3595357103 100644 --- a/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx +++ b/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx @@ -1,5 +1,5 @@ import PropTypes from "prop-types"; -import { useEffect, useState } from "react"; +import { useState } from "react"; import { Button } from "@/components/ui/shims/antd-button"; import { Form } from "@/components/ui/shims/antd-form"; import { Checkbox, Input, Select } from "@/components/ui/shims/antd-inputs"; @@ -68,23 +68,17 @@ function CreateNotification({ editDetails, }) { const [form] = Form.useForm(); - const [formDetails, setFormDetails] = useState(DEFAULT_FORM_DETAILS); + // Lazy initializer: this component remounts fresh every time Edit opens + // (NotificationModal renders DisplayNotifications in between), so editDetails + // is already the row being edited by this first render. Seeding formDetails + // here -- rather than via a useEffect a render later -- matters because the + // Form shim seeds its fields from `initialValues` on ITS OWN first mount + // only, same as real antd; seeding one render late meant the form always + // mounted on the blank defaults, and the saved name/url never appeared. + const [formDetails, setFormDetails] = useState( + () => editDetails ?? DEFAULT_FORM_DETAILS, + ); const [backendErrors, setBackendErrors] = useState(null); - const [resetForm, setResetForm] = useState(false); - - useEffect(() => { - if (editDetails) { - setFormDetails(editDetails); - setResetForm(true); - } - }, [editDetails]); - - useEffect(() => { - if (resetForm) { - form.resetFields(); - setResetForm(false); - } - }, [formDetails]); const handleInputChange = (changedValues, allValues) => { const nextValues = { ...formDetails, ...allValues }; From 99e2d8c6235560598bb405246dc1065800cb7764 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 19:20:30 +0530 Subject: [PATCH 21/28] UN-3494 [FIX] Trim the lazy-init comment down to the non-obvious reason Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- .../notification-modal/CreateNotification.jsx | 8 +------- 1 file changed, 1 insertion(+), 7 deletions(-) diff --git a/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx b/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx index 3595357103..44c6df5a16 100644 --- a/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx +++ b/frontend/src/components/pipelines-or-deployments/notification-modal/CreateNotification.jsx @@ -68,13 +68,7 @@ function CreateNotification({ editDetails, }) { const [form] = Form.useForm(); - // Lazy initializer: this component remounts fresh every time Edit opens - // (NotificationModal renders DisplayNotifications in between), so editDetails - // is already the row being edited by this first render. Seeding formDetails - // here -- rather than via a useEffect a render later -- matters because the - // Form shim seeds its fields from `initialValues` on ITS OWN first mount - // only, same as real antd; seeding one render late meant the form always - // mounted on the blank defaults, and the saved name/url never appeared. + // Lazy init: the Form shim seeds initialValues only on first mount. const [formDetails, setFormDetails] = useState( () => editDetails ?? DEFAULT_FORM_DETAILS, ); From 921acf72516eef3a76cdc0d16b5441fdfd373f27 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 20:24:43 +0530 Subject: [PATCH 22/28] UN-3494 [FIX] Skip the removal email when access never depended on shares Greptile P1 on membership_views.py:127-129. retained_user_ids returns None to mean access is unconditional (org-wide share, frictionless adapter) -- the guard only checked the concrete-set case, so a None result fell through and sent an incorrect access-removed email. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/membership_views.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/permissions/membership_views.py b/backend/permissions/membership_views.py index b1b9694db8..6c296beb30 100644 --- a/backend/permissions/membership_views.py +++ b/backend/permissions/membership_views.py @@ -125,7 +125,7 @@ def _notify_owner_removed(self, resource: Any, user: User, actor: Any) -> None: # way (a group, a direct share, an org-wide share, org admin), same as # the direct-share and group-revoke paths already check. retained = retained_user_ids(resource) - if retained is not None and user.pk in retained: + if retained is None or user.pk in retained: return resource_type, resource_name = ctx try: From a28b770338d667d0a09f4a68d50aa4141359cab1 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Wed, 16 Sep 2026 20:24:52 +0530 Subject: [PATCH 23/28] UN-3494 [FIX] Correct LookupDefinition's registered primary-key field id_field was "id", but LookupDefinition's actual primary key is lookup_id (a UUIDField). This failed Django's system check outright (tenant_account_v2.E001), which blocks manage.py check/migrate/test for the whole project -- found while verifying the fix above. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/tenant_account_v2/shareable_resources.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/tenant_account_v2/shareable_resources.py b/backend/tenant_account_v2/shareable_resources.py index f0e7881780..b418ade735 100644 --- a/backend/tenant_account_v2/shareable_resources.py +++ b/backend/tenant_account_v2/shareable_resources.py @@ -55,7 +55,7 @@ class ShareableResource: "LookupDefinition", "lookup_definition", "name", - "id", + "lookup_id", ), ) From ba181a8e42ac352dac3742bc482fb6b8f514a8c4 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 22 Sep 2026 15:38:20 +0530 Subject: [PATCH 24/28] UN-3494 [FIX] Address standardized review: 4 High + 6 Medium + 9 Low High: a permanent send failure no longer raises into PG redelivery (re-mailing a whole group on a lost-after-send response); send results are now propagated end to end instead of discarded, so a failed send returns non-2xx and a misconfiguration returns 200/skipped; missing multi-group fan-out test coverage added. Medium: enqueue path now skips when the notification plugin isn't loaded (pure OSS no longer writes dead queue rows); split the conflated actor-left-org vs unresolved-resource-type log into distinct metric keys; corrected the retry-timeout-budget comment (154s worst case, not 150 -- missed the inter-attempt sleep delays) and the test pinning it; stale docstrings corrected (atomic-block claim, axis-agnostic claim, ENABLE_EMAIL_NOTIFICATIONS-only claim). Low: added_user_ids response now echoes what was actually inserted, not the full request, plus a test for the concurrent-add narrowing; str(enum) -> .value on both enum serializations in the outbound payload; two more stale docstrings corrected (one-query-vs-two, service-account exception to the frictionless-adapter claim). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/resource_share_views.py | 15 ++--- .../group_notification_service.py | 63 +++++++++++++------ backend/tenant_account_v2/group_views.py | 2 +- backend/tenant_account_v2/internal_views.py | 11 +++- .../tenant_account_v2/share_notifications.py | 28 ++++++--- backend/tenant_account_v2/sharing_helpers.py | 4 +- .../test_share_notification_dispatch.py | 21 ++++++- backend/tenant_account_v2/tests.py | 44 +++++++++++++ workers/notification/tasks.py | 46 +++++++++++--- workers/tests/test_group_notification_post.py | 21 +++++-- 10 files changed, 198 insertions(+), 57 deletions(-) diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 576053ec30..5c19f733a9 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -1,9 +1,10 @@ """Shared share-management surface for resource ViewSets. -The mixin reads the sharing "axes" named in ``_SUPPORTED_SHARE_AXES``. -``shared_users`` is the direct-viewer axis, backed by VIEWER membership rows, -while ``shared_groups`` is stored polymorphically in ``ResourceGroupShare`` -(not an M2M) and routed through the sharing helpers. +The write side accepts exactly the three axes named in +``_SUPPORTED_SHARE_AXES``; the read side (``_read_axis``) only knows the two +per-recipient ones by name. ``shared_users`` is the direct-viewer axis, backed +by VIEWER membership rows, while ``shared_groups`` is stored polymorphically in +``ResourceGroupShare`` (not an M2M) and routed through the sharing helpers. """ import logging @@ -152,9 +153,9 @@ def share(self, request: Request, pk: str | None = None) -> Response: ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) - # ``_commit`` is the only atomic block on this path, so it has already - # committed — the diffs read persisted state and can never announce a - # share that rolled back. + # ``authorize_and_commit`` has already committed here: ``ATOMIC_REQUESTS`` + # is off, so this view isn't itself wrapped in a transaction, and the + # diffs below read persisted state rather than one that could roll back. resource.refresh_from_db() # Only the two per-recipient axes notify. ``shared_to_org`` is left out # deliberately: a toggle has no recipient list short of the whole org, diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 7f052d9304..9dbfa12815 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -82,35 +82,53 @@ def send_resource_shared( resource_id: str, share_action: str, revoked_at: datetime | None = None, -) -> None: +) -> bool: """Mail every current member of each group whose resource access changed. One email per group, so ``group_name`` in the template is always the group the recipient actually belongs to. ``share_action`` picks the wording, and on a revoke ``revoked_at`` bounds who counts as "current". + + Returns: + ``False`` only when a group with real recipients was actually attempted + and the plugin reported a send failure -- the caller's cue to ask for + redelivery. Every other outcome (nothing to send, plugin absent, + misconfigured) is ``True``: there is nothing a retry would fix. """ service = _service() if service is None: - return + return True actor = _get_user(organization, actor_id) shared = _load_resource(organization, resource_kind, resource_id) - if actor is None or shared.type is None: + if actor is None: + # Actor left the org between the share and the send -- routine race, + # not a bug. logger.info( + "metric=group_notification_actor_left_org_total group-notification: " + "skipping resource share for %s/%s (actor no longer in org)", + resource_kind, + resource_id, + ) + return True + if shared.type is None: + # A registered resource kind with no notification-plugin type mapping + # -- a real gap worth an operator's attention, unlike the actor case. + logger.warning( + "metric=group_notification_unresolved_resource_type_total " "group-notification: skipping resource share for %s/%s " - "(actor_found=%s resource_type=%s)", + "(resource type not registered)", resource_kind, resource_id, - actor is not None, - shared.type, ) - return + return True retained = _retained_user_ids(organization, shared.instance, share_action) if retained is None: - return + return True groups = list(_groups_to_mail(organization, group_ids, shared.instance, share_action)) recipients_by_group = _group_recipients_batch( organization, groups, retained, revoked_at ) + all_sent = True for group in groups: recipients = recipients_by_group.get(group.pk, []) logger.info( @@ -121,7 +139,9 @@ def send_resource_shared( len(recipients), ) if recipients: - _mail_group(service, group, recipients, shared, actor, share_action) + sent = _mail_group(service, group, recipients, shared, actor, share_action) + all_sent = all_sent and sent + return all_sent def send_membership_changed( @@ -131,17 +151,21 @@ def send_membership_changed( actor_id: int, membership_action: str, user_ids: Iterable[int], -) -> None: +) -> bool: """Mail the users whose membership of ``group_id`` just changed. Recipients are re-validated against ``OrganizationMember`` — this is where the offboarding race closes, for removals as well as additions: leaving a group does not remove someone from the org, so both directions validate the same way. + + Returns: + ``False`` only when there were real recipients and the plugin reported + a send failure. See :func:`send_resource_shared` for the full contract. """ service = _service() if service is None: - return + return True actor = _get_user(organization, actor_id) group = _groups_in_org(organization, [group_id]).first() if actor is None or group is None: @@ -152,7 +176,7 @@ def send_membership_changed( actor is not None, group is not None, ) - return + return True recipients = _live_member_users(organization, user_ids) logger.info( "group-notification: task=%s group_id=%s action=%s recipient_count=%d", @@ -162,8 +186,8 @@ def send_membership_changed( len(recipients), ) if not recipients: - return - service.send_group_membership_notification( + return True + return service.send_group_membership_notification( group_name=group.name, membership_action=MembershipAction(membership_action).value, recipients=recipients, @@ -270,9 +294,10 @@ def _group_recipients_batch( ) -> dict[int, list[User]]: """Live members of each of ``groups`` who did not keep access via ``retained``. - One query across every group in the fan-out rather than one per group -- - a resource shared with N groups issued N ``OrganizationMember`` queries - before this, since ``joined_before`` (a revoke's timestamp) is the same + Two queries total (one ``GroupMembership`` scan, one ``OrganizationMember`` + validation) across every group in the fan-out, rather than one pair per + group -- a resource shared with N groups issued N pairs of queries before + this, since ``joined_before`` (a revoke's timestamp) is the same cutoff for every group being mailed in one call, so the membership lookup batches cleanly. @@ -310,9 +335,9 @@ def _mail_group( shared: _SharedResource, actor: User, share_action: str, -) -> None: +) -> bool: """Send one group's copy of the resource-share email.""" - service.send_group_resource_shared_notification( + return service.send_group_resource_shared_notification( resource_type=shared.type, resource_name=shared.name, resource_id=str(shared.instance.pk), diff --git a/backend/tenant_account_v2/group_views.py b/backend/tenant_account_v2/group_views.py index e8e7091aa0..4814f883f7 100644 --- a/backend/tenant_account_v2/group_views.py +++ b/backend/tenant_account_v2/group_views.py @@ -178,7 +178,7 @@ def members(self, request: Request, pk: str | None = None) -> Response: actor=request.user, ) return Response( - {"added_user_ids": user_ids_to_add}, + {"added_user_ids": newly_added_ids}, status=status.HTTP_201_CREATED, ) diff --git a/backend/tenant_account_v2/internal_views.py b/backend/tenant_account_v2/internal_views.py index 08c752041e..434ee453e1 100644 --- a/backend/tenant_account_v2/internal_views.py +++ b/backend/tenant_account_v2/internal_views.py @@ -76,11 +76,16 @@ def post(self, request: Request) -> Response: serializer.is_valid(raise_exception=True) data = serializer.validated_data try: - send_resource_shared(organization=self._organization(), **data) + sent = send_resource_shared(organization=self._organization(), **data) except ResourceNotFoundError as exc: # Deleted between the share and the send — a retry cannot help. logger.info("group-notification: dropping resource share (%s)", exc) return Response({"status": "skipped"}, status=status.HTTP_200_OK) + if not sent: + # A group had real recipients and the plugin failed to send -- + # non-2xx so the worker's retry loop (and PG redelivery once that's + # exhausted) actually fires, instead of a lost email reporting green. + return Response({"status": "failed"}, status=status.HTTP_502_BAD_GATEWAY) return Response({"status": "success"}, status=status.HTTP_200_OK) @@ -90,7 +95,9 @@ class GroupMembershipChangedView(_GroupNotificationView): def post(self, request: Request) -> Response: serializer = GroupMembershipChangedSerializer(data=request.data) serializer.is_valid(raise_exception=True) - send_membership_changed( + sent = send_membership_changed( organization=self._organization(), **serializer.validated_data ) + if not sent: + return Response({"status": "failed"}, status=status.HTTP_502_BAD_GATEWAY) return Response({"status": "success"}, status=status.HTTP_200_OK) diff --git a/backend/tenant_account_v2/share_notifications.py b/backend/tenant_account_v2/share_notifications.py index 869dc7cab1..6b68301670 100644 --- a/backend/tenant_account_v2/share_notifications.py +++ b/backend/tenant_account_v2/share_notifications.py @@ -9,10 +9,11 @@ backend does the ORM and plugin work. Transport is the PG queue, the only one there is — the ``notifications`` queue the notification consumer polls. -A missing org or any dispatch error means no notification, never a broken -share. Like the direct-user mail in ``ResourceShareManagementMixin`` and the -co-owner mail it reuses, sending is bounded only by the cloud -``ENABLE_EMAIL_NOTIFICATIONS`` setting. +A missing org, an absent notification plugin, or any dispatch error means no +notification, never a broken share. Sending is also bounded by the cloud +``ENABLE_EMAIL_NOTIFICATIONS`` setting and by each event's own template ID +being configured -- either being unset silently skips, same as the +direct-user mail this reuses. """ from __future__ import annotations @@ -23,6 +24,7 @@ from typing import TYPE_CHECKING, Any from django.utils import timezone +from plugins import get_plugin from tenant_account_v2.shareable_resources import kind_for_instance @@ -108,8 +110,7 @@ def _notify_group_share( # viewset, so an unregistered kind means a group share landed on a # resource this feature cannot mail about. logger.warning( - "group-notification: skipping %s share for %s %s " - "(organization=%s kind=%s)", + "group-notification: skipping %s share for %s %s (organization=%s kind=%s)", share_action, type(resource).__name__, resource.pk, @@ -122,7 +123,7 @@ def _notify_group_share( "actor_id": actor.pk, "resource_kind": kind, "resource_id": str(resource.pk), - "share_action": str(share_action), + "share_action": share_action.value, "organization_id": organization_id, } if revoked_at is not None: @@ -163,7 +164,7 @@ def notify_group_membership_changed( kwargs={ "group_id": group.pk, "actor_id": actor.pk, - "membership_action": str(action), + "membership_action": action.value, "user_ids": recipients, "organization_id": organization_id, }, @@ -191,7 +192,18 @@ def _dispatch_quietly( The share or membership change has already been committed by the time this runs — losing its email is not a reason to fail the request the user made. + + Skips the enqueue entirely when the notification plugin isn't loaded (pure + OSS): the worker would just claim the row and no-op, so there is no point + writing a queue row nobody can act on. """ + if not get_plugin("notification"): + logger.debug( + "group-notification: notification plugin unavailable, skipping " + "enqueue for %s", + task_name, + ) + return try: _dispatch( task_name=task_name, diff --git a/backend/tenant_account_v2/sharing_helpers.py b/backend/tenant_account_v2/sharing_helpers.py index dc93d57ee5..c7775bb682 100644 --- a/backend/tenant_account_v2/sharing_helpers.py +++ b/backend/tenant_account_v2/sharing_helpers.py @@ -274,7 +274,9 @@ def access_survives_share_changes(resource_obj: Any) -> bool: Revoking a share removes nothing in that case, so nobody should be told it did. ``shared_to_org`` covers every org member; ``is_friction_less`` is the - adapter equivalent -- ``AdapterInstance.for_user`` admits it unconditionally. + adapter equivalent -- ``AdapterInstance.for_user`` admits it to any regular + user unconditionally (service accounts are the one exception, explicitly + excluded from frictionless adapters). """ return bool(getattr(resource_obj, "shared_to_org", False)) or bool( getattr(resource_obj, "is_friction_less", False) diff --git a/backend/tenant_account_v2/test_share_notification_dispatch.py b/backend/tenant_account_v2/test_share_notification_dispatch.py index 4dc5063a16..f711a3457d 100644 --- a/backend/tenant_account_v2/test_share_notification_dispatch.py +++ b/backend/tenant_account_v2/test_share_notification_dispatch.py @@ -2,8 +2,9 @@ ``share_notifications`` runs inside the user's share request: it builds the task payload and hands it to the transport. Nothing here touches the ORM or sends -mail, so the module is patched at its two seams — ``kind_for_instance`` and -``_dispatch`` — and these run in the rig's unit tier with no Postgres. The transport itself is covered by ``pg_queue.tests``; the +mail, so the module is patched at its three seams — ``kind_for_instance``, +``get_plugin`` and ``_dispatch`` — and these run in the rig's unit tier with no +Postgres. The transport itself is covered by ``pg_queue.tests``; the delivery side by ``ResourceShareNotificationTests`` in ``tenant_account_v2.tests``. """ @@ -28,9 +29,10 @@ def _group(pk: int) -> SimpleNamespace: @contextmanager def _seams(*, dispatch_raises: Exception | None = None): - """Patch the module's two outbound seams; yield the dispatch mock.""" + """Patch the module's three outbound seams; yield the dispatch mock.""" with ( patch.object(sn, "kind_for_instance", return_value="workflow"), + patch.object(sn, "get_plugin", return_value={"service_class": object()}), patch.object(sn, "_dispatch", side_effect=dispatch_raises) as dispatch, ): yield dispatch @@ -87,6 +89,7 @@ def test_no_groups_skips_dispatch(self): def test_unknown_resource_kind_skips_dispatch(self): with ( patch.object(sn, "kind_for_instance", return_value=None), + patch.object(sn, "get_plugin", return_value={"service_class": object()}), patch.object(sn, "_dispatch") as dispatch, ): sn.notify_resource_group_share_changed( @@ -102,6 +105,18 @@ def test_missing_organization_skips_dispatch(self): ) dispatch.assert_not_called() + def test_plugin_absent_skips_dispatch(self): + # Pure OSS: nothing would ever consume this row, so don't write it. + with ( + patch.object(sn, "kind_for_instance", return_value="workflow"), + patch.object(sn, "get_plugin", return_value=None), + patch.object(sn, "_dispatch") as dispatch, + ): + sn.notify_resource_group_share_changed( + resource=_RESOURCE, added=[_group(1)], removed=[], actor=_ACTOR + ) + dispatch.assert_not_called() + def test_dispatch_failure_never_reaches_the_caller(self): # The share has already committed — losing its email must not 500 it. with _seams(dispatch_raises=RuntimeError("queue down")): diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index 77bc47aef0..d23ddb3e67 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -379,6 +379,23 @@ def test_service_account_can_add_members(self) -> None: GroupMembership.objects.filter(group=self.group, user=self.outsider).exists() ) + def test_add_members_response_only_lists_newly_added(self) -> None: + # self.member is already in self.group (see GroupSharingTestBase); + # only self.outsider is new. The response, and the notification it + # feeds, must both narrow to the actual insert, not the request. + response = self._call( + {"post": "members"}, + "post", + self.svc, + data={"user_ids": [self.member.id, self.outsider.id]}, + pk=str(self.group.pk), + ) + self.assertEqual(response.status_code, 201) + self.assertEqual(response.data["added_user_ids"], [self.outsider.id]) + self.assertEqual( + GroupMembership.objects.filter(group=self.group, user=self.member).count(), 1 + ) + def test_service_account_can_remove_member(self) -> None: response = self._call( {"delete": "remove_member"}, @@ -558,6 +575,33 @@ def test_revoke_mails_members_although_the_share_row_is_gone(self) -> None: self.assertEqual(kwargs["resource_type"], "workflow") self.assertEqual(kwargs["resource_name"], "wf-1") + def test_multi_group_fan_out_keeps_each_group_and_its_own_members_separate( + self, + ) -> None: + # ``self.member`` is in both groups (overlapping); ``self.outsider`` is + # only in the second (disjoint). A naive union across groups would + # either merge them into one mail or mislabel which group a recipient + # actually belongs to -- both are exactly what one email per group + # exists to prevent. + other_group = OrganizationGroup.objects.create( + organization=self.org, name="Ops", created_by=self.owner + ) + GroupMembership.objects.create(group=other_group, user=self.member) + GroupMembership.objects.create(group=other_group, user=self.outsider) + set_resource_share_groups(self.workflow, [self.group.id, other_group.id]) + + self._send(group_ids=[self.group.id, other_group.id]) + + self.assertEqual( + sorted(self._mailed()), + sorted( + [ + ("Team", ["member@example.com"]), + ("Ops", ["member@example.com", "outsider@example.com"]), + ] + ), + ) + def test_group_from_another_org_is_never_mailed(self) -> None: """The foreign group's member is deliberately also an org-A member. diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 0cb826baa2..087151fb70 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -515,9 +515,12 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # Per-phase, because httpx has NO whole-request timeout: a scalar one is applied # to connect, write and read separately. The loop below is the only retry -- # transport-level retries would stack their own timeouts underneath these. -# Sum x _GROUP_NOTIFICATION_ATTEMPTS is the task's bound, and must stay under -# WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS (the heartbeat is frozen for the -# task's duration) and VT_SECONDS. Excludes DNS, which connect does not cover. +# Worst case per attempt is connect+write+read+pool = 50s. Across +# _GROUP_NOTIFICATION_ATTEMPTS attempts plus the sleep between each retry, the +# task's real bound is 3*50 + 2*_GROUP_NOTIFICATION_RETRY_DELAY = 154s, and +# must stay under WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS (the heartbeat +# is frozen for the task's duration) and VT_SECONDS. Excludes DNS, which +# connect does not cover. _GROUP_NOTIFICATION_TIMEOUT = httpx.Timeout(connect=5.0, write=10.0, read=30.0, pool=5.0) @@ -574,10 +577,11 @@ def _post_group_notification_once( def _fail_group_notification(endpoint: str, organization_id: str, error: str) -> None: - """Log and raise once the attempt budget is exhausted. + """Log and raise once a retryable failure exhausts its in-process attempts. - A duplicate email from the resulting redelivery is the only effect, since - the send path writes nothing. + The raise leaves the message on the queue for redelivery -- correct here + because the failure is transient (5xx / connection-level), so a later + attempt has a real chance of succeeding. """ logger.error( "metric=group_notification_post_failed_total endpoint=%s org_id=%s error=%s", @@ -588,15 +592,34 @@ def _fail_group_notification(endpoint: str, organization_id: str, error: str) -> raise RuntimeError(f"Group notification {endpoint} failed: {error}") +def _drop_group_notification(endpoint: str, organization_id: str, error: str) -> None: + """Log a permanent failure without raising. + + A non-retryable failure (a definitive 4xx, or a response lost after the + backend already sent the group's emails) will not succeed on redelivery -- + and since one send call mails a whole group with no per-recipient + checkpoint, redelivering it re-mails everyone who already got it. Raising + here would trade a dropped notification for a duplicated one. + """ + logger.error( + "metric=group_notification_dropped_total endpoint=%s org_id=%s error=%s", + endpoint, + organization_id, + error, + ) + + def _post_group_notification(endpoint: str, organization_id: str, payload: dict) -> None: """POST a group-notification job to the backend and insist it succeeded. - Deliberately raises on failure -- nothing tracks an unsent group email, so - a swallowed error would be a silent drop. The raise leaves the message on - the queue for redelivery, bounded by the consumer's attempt cap. + Raises only on a retryable failure -- nothing tracks an unsent group + email, so a swallowed transient error would be a silent drop, and the + queue's own redelivery is the backstop for that. A non-retryable failure + is dropped instead of raised: see :func:`_drop_group_notification`. """ url, headers = _build_group_notification_request(endpoint, organization_id) last_error = "" + retryable = True for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): succeeded, retryable, last_error = _post_group_notification_once( url, headers, payload @@ -614,7 +637,10 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) last_error, ) time.sleep(_GROUP_NOTIFICATION_RETRY_DELAY) - _fail_group_notification(endpoint, organization_id, last_error) + if retryable: + _fail_group_notification(endpoint, organization_id, last_error) + else: + _drop_group_notification(endpoint, organization_id, last_error) @worker_task(name="notify_resource_shared_with_group") diff --git a/workers/tests/test_group_notification_post.py b/workers/tests/test_group_notification_post.py index 7f99f1d1cb..8e386f80dd 100644 --- a/workers/tests/test_group_notification_post.py +++ b/workers/tests/test_group_notification_post.py @@ -26,6 +26,7 @@ import pytest from notification.tasks import ( _GROUP_NOTIFICATION_ATTEMPTS, + _GROUP_NOTIFICATION_RETRY_DELAY, _GROUP_NOTIFICATION_TIMEOUT, _post_group_notification, notify_group_membership_changed, @@ -94,10 +95,12 @@ def test_server_error_exhausts_the_attempt_cap_then_raises(self): assert len(client.calls) == _GROUP_NOTIFICATION_ATTEMPTS assert sleep.call_count == _GROUP_NOTIFICATION_ATTEMPTS - 1 - def test_client_error_is_terminal_after_one_post(self): - # A rejected payload will be rejected again; retrying only burns budget. + def test_client_error_is_dropped_without_raising(self): + # A rejected payload will be rejected again; retrying only burns + # budget, and raising would trigger queue redelivery of a message + # that can never succeed. client, raised, sleep = _run([_response(400)]) - assert isinstance(raised, RuntimeError) + assert raised is None assert len(client.calls) == 1 assert sleep.call_count == 0 @@ -115,10 +118,12 @@ def test_request_sent_outcome_unknown_is_never_re_posted(self, exc): """Each of these means the backend already has the request. Re-posting would re-mail every group that already succeeded, so the - in-process attempts must end after one post. + in-process attempts must end after one post -- and the task must not + raise either, or queue redelivery does the exact re-post this guards + against. """ client, raised, sleep = _run([exc]) - assert isinstance(raised, RuntimeError) + assert raised is None assert len(client.calls) == 1, f"{type(exc).__name__} was re-posted" assert sleep.call_count == 0 @@ -182,10 +187,14 @@ def test_task_wall_time_stays_under_the_visibility_timeout(self): VT is 300s for this worker in both deployments; the heartbeat is frozen for the task's duration, so health-stale (360s) is the other ceiling. + Includes the sleep between retries -- a per-post-only sum undercounts + the real wall time by (attempts - 1) * retry delay. """ t = _GROUP_NOTIFICATION_TIMEOUT per_post = t.connect + t.write + t.read + t.pool - worst = per_post * _GROUP_NOTIFICATION_ATTEMPTS + sleeps = (_GROUP_NOTIFICATION_ATTEMPTS - 1) * _GROUP_NOTIFICATION_RETRY_DELAY + worst = per_post * _GROUP_NOTIFICATION_ATTEMPTS + sleeps + assert worst == 154 assert worst < 300, f"worst case {worst}s exceeds the 300s visibility timeout" From 5cb6a3bc0f438dada6580abc25c54d3f25f005c5 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 22 Sep 2026 16:04:45 +0530 Subject: [PATCH 25/28] UN-3494 [FIX] Cover the another-group retained-access route on revoke _retained_user_ids checks four routes a member could still reach a resource through (another group, a direct share, ownership, org admin); only the direct-share one had a test. A member who is in both the revoked group and a second group that still has access was untested and would have silently regressed to being mailed a wrong "you lost access" notice. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/tenant_account_v2/tests.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index d23ddb3e67..e2600eb8f2 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -790,6 +790,17 @@ def test_revoke_skips_members_who_keep_access_another_way(self) -> None: # Nothing was lost — a direct VIEWER row still reaches the resource. self.service.send_group_resource_shared_notification.assert_not_called() + def test_revoke_skips_members_who_keep_access_via_another_group(self) -> None: + # ``self.member`` is in both groups; only ``self.group`` is revoked. + other_group = OrganizationGroup.objects.create( + organization=self.org, name="Ops", created_by=self.owner + ) + GroupMembership.objects.create(group=other_group, user=self.member) + set_resource_share_groups(self.workflow, [other_group.id]) + self._send(group_ids=[self.group.id], share_action=ShareAction.REVOKED.value) + # Still reaches the resource through "Ops" -- nothing was lost. + self.service.send_group_resource_shared_notification.assert_not_called() + def test_membership_change_mails_only_the_changed_users(self) -> None: send_membership_changed( organization=self.org, From 8b9f034e1698b9b21c14ec69eefe0cf52d8c29f8 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Tue, 22 Sep 2026 19:02:08 +0530 Subject: [PATCH 26/28] UN-3494 [FIX] Concurrent group sends, shared resource-type mapping, function length - group_notification_service.py: send each group's notification concurrently (ThreadPoolExecutor, capped at 10) instead of one blocking SendGrid call per group in sequence; extracted _resolve_share/_mail_all_groups so send_resource_shared stays under the repo's 30-line function limit. - notification_resource_types.py (new): single source of truth for the adapter/pipeline -> notification ResourceType mapping, now shared by group_notification_service.py, adapter_processor_v2/views.py, and pipeline_v2/views.py instead of three independently hand-maintained copies. - resource_share_views.py: documented why direct-share notifications stay synchronous (matches every other share call site) while group notifications dispatch via PGMQ (unbounded group size justifies it). Co-Authored-By: Claude Sonnet 5 --- backend/adapter_processor_v2/views.py | 13 +-- backend/permissions/resource_share_views.py | 9 +- backend/pipeline_v2/views.py | 11 +-- .../group_notification_service.py | 94 ++++++++++++------- .../notification_resource_types.py | 34 +++++++ 5 files changed, 114 insertions(+), 47 deletions(-) create mode 100644 backend/tenant_account_v2/notification_resource_types.py diff --git a/backend/adapter_processor_v2/views.py b/backend/adapter_processor_v2/views.py index cadaf684cd..8ffae4f2af 100644 --- a/backend/adapter_processor_v2/views.py +++ b/backend/adapter_processor_v2/views.py @@ -53,8 +53,6 @@ from .models import AdapterInstance, UserDefaultAdapter notification_plugin = get_plugin("notification") -if notification_plugin: - from plugins.notification.constants import ResourceType logger = logging.getLogger(__name__) @@ -155,12 +153,11 @@ class AdapterInstanceViewSet( def get_notification_resource_type(self, resource: Any) -> str | None: if not notification_plugin: return None - return { - "LLM": ResourceType.LLM.value, - "EMBEDDING": ResourceType.EMBEDDING.value, - "VECTOR_DB": ResourceType.VECTOR_DB.value, - "X2TEXT": ResourceType.X2TEXT.value, - }.get(resource.adapter_type, ResourceType.LLM.value) + from tenant_account_v2.notification_resource_types import ( + adapter_notification_type, + ) + + return adapter_notification_type(resource.adapter_type) def get_permissions(self) -> list[Any]: # Frictionless adapters: hidden from non-owners (update/retrieve), diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 5c19f733a9..40f24a8400 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -82,7 +82,14 @@ def _users_left_without_access(instance: Model, users: set[Any]) -> list[Any]: def _send_share_notification( instance: Model, context: tuple[str, str], users: set[Any], actor: Any ) -> None: - """Email users newly granted direct access. Best-effort.""" + """Email users newly granted direct access. Best-effort. + + Sent inline, matching every other direct-share call site in the codebase + (pipelines, API deployments, connectors, ...). Group shares, by contrast, + dispatch through PGMQ (see ``share_notifications.notify_resource_group_share_changed``) + because a group's member count -- and so its send time -- is unbounded in + a way a handful of direct users is not. + """ resource_type, resource_name = context try: notification_plugin["service_class"]().send_sharing_notification( diff --git a/backend/pipeline_v2/views.py b/backend/pipeline_v2/views.py index 46c5f628f2..6ac0597e0c 100644 --- a/backend/pipeline_v2/views.py +++ b/backend/pipeline_v2/views.py @@ -42,8 +42,6 @@ from pipeline_v2.serializers.sharing import SharedUserListSerializer notification_plugin = get_plugin("notification") -if notification_plugin: - from plugins.notification.constants import ResourceType logger = logging.getLogger(__name__) @@ -61,12 +59,13 @@ class PipelineViewSet( notification_resource_name_field = "pipeline_name" def get_notification_resource_type(self, resource: Any) -> str | None: - # Only ETL/TASK pipelines map to a notification ResourceType. if not notification_plugin: return None - if resource.pipeline_type in (ResourceType.ETL.value, ResourceType.TASK.value): - return resource.pipeline_type - return None + from tenant_account_v2.notification_resource_types import ( + pipeline_notification_type, + ) + + return pipeline_notification_type(resource.pipeline_type) def get_permissions(self) -> list[Any]: # Enabling or disabling is use, not configuration, so it follows diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index 9dbfa12815..ca4c6ec416 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -13,6 +13,7 @@ import logging from collections import defaultdict +from concurrent.futures import ThreadPoolExecutor from dataclasses import dataclass from typing import TYPE_CHECKING, Any @@ -27,6 +28,10 @@ OrganizationGroup, OrganizationMember, ) +from tenant_account_v2.notification_resource_types import ( + adapter_notification_type, + pipeline_notification_type, +) from tenant_account_v2.share_notifications import MembershipAction, ShareAction from tenant_account_v2.shareable_resources import ShareableResource, descriptor_for_kind @@ -38,9 +43,16 @@ notification_plugin = get_plugin("notification") +# Ceiling on concurrent per-group sends, mirroring the email plugin's own +# ``MAX_CONCURRENT_CHUNK_SENDS`` -- a resource shared with many groups would +# otherwise serialize one blocking SendGrid call per group. +_MAX_CONCURRENT_GROUP_SENDS = 10 + # OSS ``ShareableResource.kind`` → the email plugin's ``ResourceType`` value. # Deliberately plain strings: OSS must not import a cloud-only enum. Not a 1:1 -# rename — pipelines and adapters resolve from the instance below. +# rename — pipelines and adapters resolve via the shared helpers below instead, +# the same ones the direct-share viewsets use, so a new adapter/pipeline type +# only needs registering once. _STATIC_RESOURCE_TYPES = { "workflow": "workflow", "api_deployment": "api", @@ -49,15 +61,6 @@ "agentic_project": "agentic_project", "lookup": "lookup", } -_ADAPTER_RESOURCE_TYPES = { - "LLM": "llm", - "EMBEDDING": "embedding", - "VECTOR_DB": "vector_db", - "X2TEXT": "x2text", -} -# Only ETL/TASK pipelines map to a notification resource type; the plugin -# compares against these exact (uppercase) values. -_PIPELINE_RESOURCE_TYPES = frozenset({"ETL", "TASK"}) class ResourceNotFoundError(Exception): @@ -98,6 +101,26 @@ def send_resource_shared( service = _service() if service is None: return True + resolved = _resolve_share(organization, actor_id, resource_kind, resource_id) + if resolved is None: + return True + actor, shared = resolved + retained = _retained_user_ids(organization, shared.instance, share_action) + if retained is None: + return True + groups = list(_groups_to_mail(organization, group_ids, shared.instance, share_action)) + recipients_by_group = _group_recipients_batch( + organization, groups, retained, revoked_at + ) + return _mail_all_groups( + service, groups, recipients_by_group, shared, actor, share_action + ) + + +def _resolve_share( + organization: Organization, actor_id: int, resource_kind: str, resource_id: str +) -> tuple[User, _SharedResource] | None: + """The actor and resolved resource, or ``None`` to skip (already logged).""" actor = _get_user(organization, actor_id) shared = _load_resource(organization, resource_kind, resource_id) if actor is None: @@ -109,7 +132,7 @@ def send_resource_shared( resource_kind, resource_id, ) - return True + return None if shared.type is None: # A registered resource kind with no notification-plugin type mapping # -- a real gap worth an operator's attention, unlike the actor case. @@ -120,28 +143,39 @@ def send_resource_shared( resource_kind, resource_id, ) - return True - retained = _retained_user_ids(organization, shared.instance, share_action) - if retained is None: - return True - groups = list(_groups_to_mail(organization, group_ids, shared.instance, share_action)) - recipients_by_group = _group_recipients_batch( - organization, groups, retained, revoked_at - ) - all_sent = True + return None + return actor, shared + + +def _mail_all_groups( + service: Any, + groups: list[OrganizationGroup], + recipients_by_group: dict[int, list[User]], + shared: _SharedResource, + actor: User, + share_action: str, +) -> bool: + """Send each group's copy concurrently; ``False`` if any real send failed.""" + to_mail = [g for g in groups if recipients_by_group.get(g.pk)] for group in groups: - recipients = recipients_by_group.get(group.pk, []) logger.info( "group-notification: task=notify_resource_shared_with_group " "group_id=%s action=%s recipient_count=%d", group.pk, share_action, - len(recipients), + len(recipients_by_group.get(group.pk, [])), ) - if recipients: - sent = _mail_group(service, group, recipients, shared, actor, share_action) - all_sent = all_sent and sent - return all_sent + if not to_mail: + return True + + def _send(group: OrganizationGroup) -> bool: + return _mail_group( + service, group, recipients_by_group[group.pk], shared, actor, share_action + ) + + workers = min(len(to_mail), _MAX_CONCURRENT_GROUP_SENDS) + with ThreadPoolExecutor(max_workers=workers) as pool: + return all(pool.map(_send, to_mail)) def send_membership_changed( @@ -417,11 +451,7 @@ def _resource_type_for(descriptor: ShareableResource, resource: Any) -> str | No that is neither ETL nor TASK) — the caller skips rather than guessing. """ if descriptor.kind == "pipeline": - pipeline_type = getattr(resource, "pipeline_type", None) - return pipeline_type if pipeline_type in _PIPELINE_RESOURCE_TYPES else None + return pipeline_notification_type(getattr(resource, "pipeline_type", None)) if descriptor.kind == "adapter_instance": - # Unknown adapter types fall back to ``llm``, matching the co-owner - # path's override — an OCR adapter shared with a group should not - # silently send nothing when sharing it with a co-owner mails fine. - return _ADAPTER_RESOURCE_TYPES.get(str(resource.adapter_type or ""), "llm") + return adapter_notification_type(str(resource.adapter_type or "")) return _STATIC_RESOURCE_TYPES.get(descriptor.kind) diff --git a/backend/tenant_account_v2/notification_resource_types.py b/backend/tenant_account_v2/notification_resource_types.py new file mode 100644 index 0000000000..ef4a32ae49 --- /dev/null +++ b/backend/tenant_account_v2/notification_resource_types.py @@ -0,0 +1,34 @@ +"""Resource-type mapping shared between the direct-share and group-share +notification paths, so a new adapter or pipeline type only needs registering +once. +""" + +from __future__ import annotations + + +def adapter_notification_type(adapter_type: str) -> str: + """Map an ``AdapterInstance.adapter_type`` to the plugin's ``ResourceType``. + + Unknown types fall back to ``LLM`` so a newly added adapter kind still + mails rather than silently going quiet. + """ + from plugins.notification.constants import ResourceType + + return { + "LLM": ResourceType.LLM.value, + "EMBEDDING": ResourceType.EMBEDDING.value, + "VECTOR_DB": ResourceType.VECTOR_DB.value, + "X2TEXT": ResourceType.X2TEXT.value, + }.get(adapter_type, ResourceType.LLM.value) + + +def pipeline_notification_type(pipeline_type: str | None) -> str | None: + """Map a ``Pipeline.pipeline_type`` to the plugin's ``ResourceType``. + + Only ETL/TASK pipelines are notifiable; anything else returns ``None``. + """ + from plugins.notification.constants import ResourceType + + if pipeline_type in (ResourceType.ETL.value, ResourceType.TASK.value): + return pipeline_type + return None From 9f8fd5e155930cf067e6aa156661a9554e1041c0 Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Thu, 24 Sep 2026 14:16:12 +0530 Subject: [PATCH 27/28] UN-3494 [FIX] Tri-state send result, fix the retry-loop live on dev, and the rest of ali's FOLLOWUP Critical (both): the 502-on-False contract collapsed four different plugin outcomes -- unset template, ENABLE_EMAIL_NOTIFICATIONS=false, bad input, and a genuine SendGrid failure -- into one signal, and the worker retries any 502 to its attempt cap. Every non-delivery condition was looping forever; confirmed live on dev, where ENABLE_EMAIL_NOTIFICATIONS is false today. Fixed with a tri-state result (True=sent, None=skipped/no-retry-needed, False=genuine failure) threaded from the plugin boundary through to the view's status code. Only an explicit False now returns 502. High: ThreadPoolExecutor.map wrapped in all() cancelled pending futures on the first False -- groups past the concurrency cap silently never got mailed. Fixed by materializing the full result list before aggregating. A partial group failure (some sent, some didn't) is deliberately not retried either -- redelivery would re-mail the groups that already succeeded, the exact duplication the retry split exists to prevent; the loss is logged instead. The ORM query resolving a resource's organization ran lazily inside pool threads with no connection cleanup (up to 10 leaked connections per request) -- fixed by populating the FK cache with the organization already in scope, before entering the pool. Also: cross-repo Critical from the first round reopened -- the deleted axis-diff shim was checked against the PR branches (correctly absent) instead of cloud's origin/main, where the still-stale AgenticProjectViewSet.partial_update calls it. Restored share_axes / snapshot_share_axes / diff_share_axes / AxisDiff as a temporary, explicitly deprecated shim (matches the original always-empty-diff contract exactly, pinned by a new test) so cloud's stale main doesn't AttributeError once this branch's mixin lands -- delete it once cloud #1698 merges. Medium/Low: drop-path now logs the payload (a dropped revoke is compliance-visible, not silent); explicit invariant + comment on the retry-loop's seed value; corrected the timeout-budget comment's "real bound" framing to "nominal" (a trickling response defeats per-phase read timeouts); stale docstring/comment fixes (env-knob framing, resource- type-mapping duplication, _live_member_users' undocumented email filter); notification-narrowing test for the add-members endpoint. 11 new/extended tests. 1719 backend tests (1714 pass, 5 pre-existing environment-gap errors unrelated to this diff, 8 skipped), 1522 worker tests (1 skipped) -- zero regressions. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- backend/permissions/resource_share_views.py | 54 ++++++++++- .../tests/test_owner_management.py | 40 +++++++- .../group_notification_service.py | 69 +++++++++++--- .../tenant_account_v2/test_internal_views.py | 93 +++++++++++++++++++ backend/tenant_account_v2/tests.py | 62 +++++++++++-- workers/notification/tasks.py | 30 ++++-- workers/tests/test_group_notification_post.py | 10 ++ 7 files changed, 320 insertions(+), 38 deletions(-) create mode 100644 backend/tenant_account_v2/test_internal_views.py diff --git a/backend/permissions/resource_share_views.py b/backend/permissions/resource_share_views.py index 40f24a8400..a042f5d782 100644 --- a/backend/permissions/resource_share_views.py +++ b/backend/permissions/resource_share_views.py @@ -8,7 +8,8 @@ """ import logging -from typing import Any +from dataclasses import dataclass, field +from typing import Any, ClassVar from django.db.models import Model from plugins import get_plugin @@ -122,9 +123,53 @@ def _send_revoke_notification( logger.exception("Failed to send access-removed notification for %s", instance.pk) +@dataclass +class AxisDiff: + """DEPRECATED shim -- see ``ResourceShareManagementMixin.share_axes`` below.""" + + before: set[Any] = field(default_factory=set) + after: set[Any] = field(default_factory=set) + + @property + def added(self) -> set[Any]: + return self.after - self.before + + @property + def removed(self) -> set[Any]: + return self.before - self.after + + class ResourceShareManagementMixin: """Adds the shared share-management surface to a resource ViewSet.""" + # DEPRECATED, temporary: ``share_axes``, ``snapshot_share_axes`` and + # ``diff_share_axes`` (with ``AxisDiff`` above) were removed on this branch + # -- the PATCH-based sharing path they backed is dead, its diff is always + # empty (see 999e443b4). Restored here only because cloud's + # ``AgenticProjectViewSet.partial_update`` on ``origin/main`` still calls + # them, and OSS merges before cloud (Zipstack/unstract-cloud#1698 carries + # the real removal). Delete this whole block, and this comment, once #1698 + # merges -- at that point nothing on cloud main calls it anymore. + share_axes: ClassVar[tuple[str, ...]] = ("shared_users", "shared_groups") + + def snapshot_share_axes(self, instance: Model) -> dict[str, set[Any]]: + """DEPRECATED shim. See the block comment above ``share_axes``.""" + return {axis: self._read_axis(instance, axis) for axis in self.share_axes} + + def diff_share_axes( + self, + instance: Model, + before: dict[str, set[Any]], + request_data: dict[str, Any], + ) -> dict[str, AxisDiff]: + """DEPRECATED shim. See the block comment above ``share_axes``.""" + instance.refresh_from_db() + return { + axis: AxisDiff(before=before[axis], after=self._read_axis(instance, axis)) + for axis in self.share_axes + if axis in request_data + } + @action(detail=True, methods=["post"], url_path="share") def share(self, request: Request, pk: str | None = None) -> Response: """Apply a replace-style share state for the resource. @@ -160,9 +205,10 @@ def share(self, request: Request, pk: str | None = None) -> Response: ShareAuthorizationService.authorize_and_commit( actor=request.user, resource=resource, desired=desired ) - # ``authorize_and_commit`` has already committed here: ``ATOMIC_REQUESTS`` - # is off, so this view isn't itself wrapped in a transaction, and the - # diffs below read persisted state rather than one that could roll back. + # ``authorize_and_commit`` has already committed here, on the current + # deployment: ``ATOMIC_REQUESTS`` is a settings knob, currently off, so + # this view isn't wrapped in a transaction and the diffs below read + # persisted state. Flipping that knob would flip this premise too. resource.refresh_from_db() # Only the two per-recipient axes notify. ``shared_to_org`` is left out # deliberately: a toggle has no recipient list short of the whole org, diff --git a/backend/permissions/tests/test_owner_management.py b/backend/permissions/tests/test_owner_management.py index 7e016a56da..b5804799f7 100644 --- a/backend/permissions/tests/test_owner_management.py +++ b/backend/permissions/tests/test_owner_management.py @@ -16,7 +16,7 @@ import pytest from account_v2.models import User from django.test import TestCase -from permissions.roles import ResourceRole +from prompt_studio.permission import ParentToolAccess from rest_framework import status from rest_framework.parsers import JSONParser from rest_framework.request import Request as DRFRequest @@ -27,7 +27,7 @@ from workflow_manager.workflow_v2.views import WorkflowViewSet from permissions.membership_serializers import AddOwnerSerializer -from prompt_studio.permission import ParentToolAccess +from permissions.roles import ResourceRole from permissions.tests.base import ( RESOURCE_SPECS, CoOwnerOrgTestMixin, @@ -386,9 +386,7 @@ def test_parent_tool_viewer_allowed_outsider_denied(self) -> None: def test_null_parent_falls_back_to_object_creator(self) -> None: # No parent tool → access derives from the object's own ``created_by``. - orphan = SimpleNamespace( - prompt_studio_tool=None, created_by_id=self.owner.pk - ) + orphan = SimpleNamespace(prompt_studio_tool=None, created_by_id=self.owner.pk) self.assertTrue(self._perm(self.owner, orphan)) self.assertFalse(self._perm(self.coowner, orphan)) @@ -673,3 +671,35 @@ def test_import_path_grants_creator_ownership(self) -> None: ) self.assertTrue(tool.is_owner(self.coowner)) self.assertIn(tool, CustomTool.objects.for_user(self.coowner)) + + +class DeprecatedAxisShimTests(CoOwnerOrgTestMixin, TestCase): + """The temporary ``share_axes``/``snapshot_share_axes``/``diff_share_axes`` + shim (see the block comment in ``resource_share_views.py``) -- restored + only so cloud's still-stale ``origin/main`` ``AgenticProjectViewSet`` + doesn't ``AttributeError`` once this branch merges and deletes the real + thing on OSS main. Delete this test alongside that block, once + Zipstack/unstract-cloud#1698 merges. + """ + + def setUp(self) -> None: + self._seed_org() + self.workflow = Workflow.objects.create( + workflow_name="wf-shim", organization=self.org, created_by=self.owner + ) + self.workflow.memberships.create(user=self.owner, role=ResourceRole.OWNER) + + def test_snapshot_then_diff_reproduces_the_original_always_empty_shape(self) -> None: + # Matches the pre-removal contract exactly: the PATCH path this shim + # exists for has no way to actually change shared_users/shared_groups + # (see 999e443b4), so the diff for either axis is always empty. + view = WorkflowViewSet() + before = view.snapshot_share_axes(self.workflow) + self.assertEqual(before, {"shared_users": set(), "shared_groups": set()}) + diff = view.diff_share_axes( + self.workflow, before, {"shared_users": [], "shared_groups": []} + ) + self.assertEqual(set(diff), {"shared_users", "shared_groups"}) + for axis_diff in diff.values(): + self.assertEqual(axis_diff.added, set()) + self.assertEqual(axis_diff.removed, set()) diff --git a/backend/tenant_account_v2/group_notification_service.py b/backend/tenant_account_v2/group_notification_service.py index ca4c6ec416..c95929ca07 100644 --- a/backend/tenant_account_v2/group_notification_service.py +++ b/backend/tenant_account_v2/group_notification_service.py @@ -48,11 +48,16 @@ # otherwise serialize one blocking SendGrid call per group. _MAX_CONCURRENT_GROUP_SENDS = 10 -# OSS ``ShareableResource.kind`` → the email plugin's ``ResourceType`` value. -# Deliberately plain strings: OSS must not import a cloud-only enum. Not a 1:1 -# rename — pipelines and adapters resolve via the shared helpers below instead, -# the same ones the direct-share viewsets use, so a new adapter/pipeline type -# only needs registering once. +# OSS ``ShareableResource.kind`` → the email plugin's ``ResourceType`` value, +# for the 6 kinds that are a plain 1:1 rename. Pipelines and adapters resolve +# via ``notification_resource_types`` instead, the same helpers the +# direct-share viewsets use, so a new adapter/pipeline type only needs +# registering once. These 6 are still a second, hand-maintained copy of what +# each ViewSet's own ``get_notification_resource_type`` already states -- +# unifying them the same way is a larger change than this one, tracked +# separately. Plain strings here, not the cloud enum: this dict is only ever +# read once the plugin is confirmed loaded (see ``_service()``), but nothing +# enforces that path if a future caller reached it another way. _STATIC_RESOURCE_TYPES = { "workflow": "workflow", "api_deployment": "api", @@ -155,7 +160,16 @@ def _mail_all_groups( actor: User, share_action: str, ) -> bool: - """Send each group's copy concurrently; ``False`` if any real send failed.""" + """Send each group's copy concurrently. + + Returns ``False`` (ask for redelivery) only when every attempted group + genuinely failed to send. A skipped group (the plugin's tri-state + ``None`` -- unconfigured, disabled, bad input) never counts as a failure. + A *partial* failure -- some groups sent, others didn't -- is deliberately + not retried either: redelivery would re-mail the groups that already + succeeded, which is the exact duplication the worker's own retry + classification exists to avoid. The lost group is logged instead. + """ to_mail = [g for g in groups if recipients_by_group.get(g.pk)] for group in groups: logger.info( @@ -168,14 +182,29 @@ def _mail_all_groups( if not to_mail: return True - def _send(group: OrganizationGroup) -> bool: + def _send(group: OrganizationGroup) -> bool | None: return _mail_group( service, group, recipients_by_group[group.pk], shared, actor, share_action ) workers = min(len(to_mail), _MAX_CONCURRENT_GROUP_SENDS) with ThreadPoolExecutor(max_workers=workers) as pool: - return all(pool.map(_send, to_mail)) + # list() first: pool.map returns a lazy generator, and all() stopping + # at the first False would cancel every pending future past it -- + # groups later in the batch would silently never be mailed at all. + results = list(pool.map(_send, to_mail)) + failed = [g.pk for g, r in zip(to_mail, results, strict=True) if r is False] + if not failed: + return True + if len(failed) == len(to_mail): + return False + logger.error( + "metric=group_notification_partial_failure_total failed_group_ids=%s " + "of %d attempted", + failed, + len(to_mail), + ) + return True def send_membership_changed( @@ -221,13 +250,16 @@ def send_membership_changed( ) if not recipients: return True - return service.send_group_membership_notification( + result = service.send_group_membership_notification( group_name=group.name, membership_action=MembershipAction(membership_action).value, recipients=recipients, actor=actor, organization=organization, ) + # Tri-state from the plugin: None (skipped -- unconfigured, disabled, bad + # input) is not a failure, only an explicit False is. + return result is not False def _service() -> Any | None: @@ -369,8 +401,13 @@ def _mail_group( shared: _SharedResource, actor: User, share_action: str, -) -> bool: - """Send one group's copy of the resource-share email.""" +) -> bool | None: + """Send one group's copy of the resource-share email. + + Passes through the plugin's tri-state result -- see + :func:`_mail_all_groups` for how ``None`` (skipped) is distinguished + from ``False`` (genuinely failed). + """ return service.send_group_resource_shared_notification( resource_type=shared.type, resource_name=shared.name, @@ -395,7 +432,10 @@ def _groups_in_org( def _live_member_users(organization: Organization, user_ids: Iterable[int]) -> list[User]: """Users from ``user_ids`` who are still live members of ``organization``. - Service accounts are excluded, matching ``compute_effective_members``. + Service accounts are excluded, matching ``compute_effective_members``. Also + drops anyone with a falsy ``email`` -- silently, since this list decides + who has real recipients, and that in turn decides whether a group is + attempted at all (and so whether a 502 can ever fire for it). """ requested = list(user_ids) memberships = OrganizationMember.objects.filter( @@ -440,6 +480,11 @@ def _load_resource( ).first() if resource is None: raise ResourceNotFoundError(f"{kind} {resource_id} not found in organization") + # Populate the FK cache with the instance we already hold: the mail send + # (``resource_instance.organization``) runs inside a pool thread, and a + # lazy query there opens a connection ``close_old_connections`` never + # cleans up (that hook only runs on the request thread). + resource.organization = organization name = getattr(resource, descriptor.name_field, "") or "" return _SharedResource(resource, name, _resource_type_for(descriptor, resource)) diff --git a/backend/tenant_account_v2/test_internal_views.py b/backend/tenant_account_v2/test_internal_views.py new file mode 100644 index 0000000000..b936bf097b --- /dev/null +++ b/backend/tenant_account_v2/test_internal_views.py @@ -0,0 +1,93 @@ +"""View-level tests for the group-notification internal endpoints. + +The send-side logic (who gets mailed, the tri-state plugin contract) is +covered in ``ResourceShareNotificationTests`` in ``tests.py``. These pin the +one thing that lives only here: the status-code mapping from that result to +an HTTP response, since that mapping is what the worker's retry decision +actually reads. ``send_resource_shared`` / ``send_membership_changed`` are +patched directly so these don't need a full group/resource DB setup to +exercise the view in isolation. +""" + +from unittest.mock import patch + +from account_v2.models import Organization +from django.test import RequestFactory, TestCase +from rest_framework import status +from utils.user_context import UserContext + +from tenant_account_v2.internal_views import ( + GroupMembershipChangedView, + ResourceSharedWithGroupView, +) + +_SHARE_PAYLOAD = { + "group_ids": [1], + "actor_id": 1, + "resource_kind": "workflow", + "resource_id": "wf-1", + "share_action": "shared", + "revoked_at": None, +} + +_MEMBERSHIP_PAYLOAD = { + "group_id": 1, + "actor_id": 1, + "membership_action": "added", + "user_ids": [1], +} + + +class _InternalViewTestBase(TestCase): + def setUp(self) -> None: + self.org = Organization.objects.create( + name="org-views", display_name="Org Views", organization_id="org-views" + ) + UserContext.set_organization_identifier(self.org.organization_id) + self.addCleanup(UserContext.set_organization_identifier, None) + + @staticmethod + def _post(view_cls, data: dict): + request = RequestFactory().post( + "/internal/", data=data, content_type="application/json" + ) + return view_cls.as_view()(request) + + +class ResourceSharedWithGroupViewTests(_InternalViewTestBase): + def test_sent_returns_200(self) -> None: + with patch( + "tenant_account_v2.internal_views.send_resource_shared", return_value=True + ): + response = self._post(ResourceSharedWithGroupView, _SHARE_PAYLOAD) + self.assertEqual(response.status_code, status.HTTP_200_OK) + self.assertEqual(response.data["status"], "success") + + def test_genuine_failure_returns_502(self) -> None: + # This is the branch the Critical hinged on: before the tri-state + # fix, every skip (unconfigured template, disabled notifications, + # bad input) collapsed into this same False -- retrying forever a + # condition no retry could fix. + with patch( + "tenant_account_v2.internal_views.send_resource_shared", return_value=False + ): + response = self._post(ResourceSharedWithGroupView, _SHARE_PAYLOAD) + self.assertEqual(response.status_code, status.HTTP_502_BAD_GATEWAY) + self.assertEqual(response.data["status"], "failed") + + +class GroupMembershipChangedViewTests(_InternalViewTestBase): + def test_sent_returns_200(self) -> None: + with patch( + "tenant_account_v2.internal_views.send_membership_changed", return_value=True + ): + response = self._post(GroupMembershipChangedView, _MEMBERSHIP_PAYLOAD) + self.assertEqual(response.status_code, status.HTTP_200_OK) + + def test_genuine_failure_returns_502(self) -> None: + with patch( + "tenant_account_v2.internal_views.send_membership_changed", + return_value=False, + ): + response = self._post(GroupMembershipChangedView, _MEMBERSHIP_PAYLOAD) + self.assertEqual(response.status_code, status.HTTP_502_BAD_GATEWAY) diff --git a/backend/tenant_account_v2/tests.py b/backend/tenant_account_v2/tests.py index e2600eb8f2..19f3d0ca6a 100644 --- a/backend/tenant_account_v2/tests.py +++ b/backend/tenant_account_v2/tests.py @@ -383,18 +383,27 @@ def test_add_members_response_only_lists_newly_added(self) -> None: # self.member is already in self.group (see GroupSharingTestBase); # only self.outsider is new. The response, and the notification it # feeds, must both narrow to the actual insert, not the request. - response = self._call( - {"post": "members"}, - "post", - self.svc, - data={"user_ids": [self.member.id, self.outsider.id]}, - pk=str(self.group.pk), - ) + with patch( + "tenant_account_v2.group_views.notify_group_membership_changed" + ) as notify: + response = self._call( + {"post": "members"}, + "post", + self.svc, + data={"user_ids": [self.member.id, self.outsider.id]}, + pk=str(self.group.pk), + ) self.assertEqual(response.status_code, 201) self.assertEqual(response.data["added_user_ids"], [self.outsider.id]) self.assertEqual( GroupMembership.objects.filter(group=self.group, user=self.member).count(), 1 ) + # The regression this guards against: passing the full request list + # (including the already-a-member id) would still pass the earlier + # assertions above -- only this call proves the notification itself + # was narrowed, not just the DB write and the response. + notify.assert_called_once() + self.assertEqual(notify.call_args.kwargs["user_ids"], [self.outsider.id]) def test_service_account_can_remove_member(self) -> None: response = self._call( @@ -534,8 +543,8 @@ def _send( group_ids: list[int], share_action: str = ShareAction.SHARED.value, revoked_at=None, - ) -> None: - send_resource_shared( + ) -> bool: + return send_resource_shared( organization=self.org, group_ids=group_ids, actor_id=self.owner.pk, @@ -602,6 +611,41 @@ def test_multi_group_fan_out_keeps_each_group_and_its_own_members_separate( ), ) + def test_plugin_skip_result_is_not_a_failure(self) -> None: + # The plugin's tri-state None means "skipped" (unconfigured, disabled, + # bad input) -- never a reason to ask for redelivery. + set_resource_share_groups(self.workflow, [self.group.id]) + self.service.send_group_resource_shared_notification.return_value = None + self.assertTrue(self._send(group_ids=[self.group.id])) + + def test_all_groups_failing_asks_for_redelivery(self) -> None: + other_group = OrganizationGroup.objects.create( + organization=self.org, name="Ops", created_by=self.owner + ) + GroupMembership.objects.create(group=other_group, user=self.outsider) + set_resource_share_groups(self.workflow, [self.group.id, other_group.id]) + self.service.send_group_resource_shared_notification.return_value = False + self.assertFalse(self._send(group_ids=[self.group.id, other_group.id])) + + def test_partial_group_failure_is_not_retried_and_every_group_is_attempted( + self, + ) -> None: + # A retry would re-mail the group that already succeeded -- accept the + # partial loss instead. Every group must still be attempted, not just + # the ones before the first failure: ThreadPoolExecutor.map's pending + # futures must not be cancelled by an early result. + other_group = OrganizationGroup.objects.create( + organization=self.org, name="Ops", created_by=self.owner + ) + GroupMembership.objects.create(group=other_group, user=self.outsider) + set_resource_share_groups(self.workflow, [self.group.id, other_group.id]) + self.service.send_group_resource_shared_notification.side_effect = [False, True] + result = self._send(group_ids=[self.group.id, other_group.id]) + self.assertTrue(result) + self.assertEqual( + self.service.send_group_resource_shared_notification.call_count, 2 + ) + def test_group_from_another_org_is_never_mailed(self) -> None: """The foreign group's member is deliberately also an org-A member. diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 087151fb70..979c42a544 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -516,11 +516,13 @@ def priority_notification(notification_type: str, **kwargs: Any) -> dict[str, An # to connect, write and read separately. The loop below is the only retry -- # transport-level retries would stack their own timeouts underneath these. # Worst case per attempt is connect+write+read+pool = 50s. Across -# _GROUP_NOTIFICATION_ATTEMPTS attempts plus the sleep between each retry, the -# task's real bound is 3*50 + 2*_GROUP_NOTIFICATION_RETRY_DELAY = 154s, and -# must stay under WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS (the heartbeat -# is frozen for the task's duration) and VT_SECONDS. Excludes DNS, which -# connect does not cover. +# _GROUP_NOTIFICATION_ATTEMPTS attempts plus the sleep between each retry, +# that's a NOMINAL budget of 3*50 + 2*_GROUP_NOTIFICATION_RETRY_DELAY = 154s -- +# not a hard bound, since ``read`` times out per socket read, not on the +# total: a response trickling in under 30s per chunk runs past this +# regardless. Sized to stay under WORKER_PG_QUEUE_CONSUMER_HEALTH_STALE_SECONDS +# (the heartbeat is frozen for the task's duration) and VT_SECONDS in the +# common case. Excludes DNS, which connect does not cover. _GROUP_NOTIFICATION_TIMEOUT = httpx.Timeout(connect=5.0, write=10.0, read=30.0, pool=5.0) @@ -592,7 +594,9 @@ def _fail_group_notification(endpoint: str, organization_id: str, error: str) -> raise RuntimeError(f"Group notification {endpoint} failed: {error}") -def _drop_group_notification(endpoint: str, organization_id: str, error: str) -> None: +def _drop_group_notification( + endpoint: str, organization_id: str, error: str, payload: dict +) -> None: """Log a permanent failure without raising. A non-retryable failure (a definitive 4xx, or a response lost after the @@ -600,12 +604,18 @@ def _drop_group_notification(endpoint: str, organization_id: str, error: str) -> and since one send call mails a whole group with no per-recipient checkpoint, redelivering it re-mails everyone who already got it. Raising here would trade a dropped notification for a duplicated one. + + The message is acked and deleted once this returns -- nothing else records + what was lost, so the payload goes in the log line (a dropped *revoke* is + compliance-visible, not just an inconvenience). """ logger.error( - "metric=group_notification_dropped_total endpoint=%s org_id=%s error=%s", + "metric=group_notification_dropped_total endpoint=%s org_id=%s error=%s " + "payload=%s", endpoint, organization_id, error, + payload, ) @@ -619,6 +629,10 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) """ url, headers = _build_group_notification_request(endpoint, organization_id) last_error = "" + # Seeded True so a zero-iteration loop would fail loud via + # _fail_group_notification (an empty last_error) rather than silently drop + # -- moot today since _GROUP_NOTIFICATION_ATTEMPTS is a fixed positive + # constant, but this is the safer default if that ever changed. retryable = True for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): succeeded, retryable, last_error = _post_group_notification_once( @@ -640,7 +654,7 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) if retryable: _fail_group_notification(endpoint, organization_id, last_error) else: - _drop_group_notification(endpoint, organization_id, last_error) + _drop_group_notification(endpoint, organization_id, last_error, payload) @worker_task(name="notify_resource_shared_with_group") diff --git a/workers/tests/test_group_notification_post.py b/workers/tests/test_group_notification_post.py index 8e386f80dd..22f10be23c 100644 --- a/workers/tests/test_group_notification_post.py +++ b/workers/tests/test_group_notification_post.py @@ -146,6 +146,16 @@ def test_recovers_when_a_later_attempt_succeeds(self): assert raised is None assert len(client.calls) == 2 + def test_retryable_then_terminal_stops_and_drops_not_raises(self): + # A transient 503 retries once, then a 400 on that retry is terminal -- + # the loop must stop there (not spend the 3rd attempt) and the final + # classification (400 -> not retryable) decides drop-without-raising, + # not the classification of the earlier, already-superseded attempt. + client, raised, sleep = _run([_response(503), _response(400)]) + assert raised is None + assert len(client.calls) == 2 + assert sleep.call_count == 1 + class TestRequestShape: def test_missing_credentials_raise_before_any_post(self): From b75e8941c948e3804f96ac029cf0f7a9a0a5872d Mon Sep 17 00:00:00 2001 From: kirtimanmishrazipstack Date: Thu, 24 Sep 2026 14:26:27 +0530 Subject: [PATCH 28/28] UN-3494 [FIX] Close out ali's remaining Low/Medium round-2 findings on the notification worker - retryable=True -> False as the loop's zero-iteration seed: a misconfigured attempt-cap of 0 must drop, not raise into an endless redelivery loop that never actually attempts a POST. - Rename the wall-time consistency test off a name that promised a measured guarantee it never checked, drop the brittle ==154 pin in favor of the two ceilings (VT 300s, health-stale 360s) it exists to protect. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018KZLGSa3oWxgVJdFqvRUQX --- workers/notification/tasks.py | 10 +++++----- workers/tests/test_group_notification_post.py | 10 ++++++---- 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/workers/notification/tasks.py b/workers/notification/tasks.py index 979c42a544..380e4f5dcb 100644 --- a/workers/notification/tasks.py +++ b/workers/notification/tasks.py @@ -629,11 +629,11 @@ def _post_group_notification(endpoint: str, organization_id: str, payload: dict) """ url, headers = _build_group_notification_request(endpoint, organization_id) last_error = "" - # Seeded True so a zero-iteration loop would fail loud via - # _fail_group_notification (an empty last_error) rather than silently drop - # -- moot today since _GROUP_NOTIFICATION_ATTEMPTS is a fixed positive - # constant, but this is the safer default if that ever changed. - retryable = True + # Seeded False: a zero-iteration loop (only possible if + # _GROUP_NOTIFICATION_ATTEMPTS were ever misconfigured to <= 0) means + # nothing was ever attempted, so drop rather than raise -- raising here + # would redeliver forever with no attempt ever being made. + retryable = False for attempt in range(1, _GROUP_NOTIFICATION_ATTEMPTS + 1): succeeded, retryable, last_error = _post_group_notification_once( url, headers, payload diff --git a/workers/tests/test_group_notification_post.py b/workers/tests/test_group_notification_post.py index 22f10be23c..4c04365b71 100644 --- a/workers/tests/test_group_notification_post.py +++ b/workers/tests/test_group_notification_post.py @@ -192,20 +192,22 @@ def test_timeout_is_per_phase_not_scalar(self): assert timeout is _GROUP_NOTIFICATION_TIMEOUT assert timeout.connect < timeout.read - def test_task_wall_time_stays_under_the_visibility_timeout(self): - """The bound the chart and compose comments defer to. + def test_timeout_constants_sum_under_the_visibility_and_health_stale_ceilings(self): + """Consistency check on the constants, not a measurement of a real POST. VT is 300s for this worker in both deployments; the heartbeat is frozen for the task's duration, so health-stale (360s) is the other ceiling. Includes the sleep between retries -- a per-post-only sum undercounts - the real wall time by (attempts - 1) * retry delay. + the real wall time by (attempts - 1) * retry delay. Nothing here times + an actual request -- this only proves the constants are still + consistent with each other, not that a real POST stays under budget. """ t = _GROUP_NOTIFICATION_TIMEOUT per_post = t.connect + t.write + t.read + t.pool sleeps = (_GROUP_NOTIFICATION_ATTEMPTS - 1) * _GROUP_NOTIFICATION_RETRY_DELAY worst = per_post * _GROUP_NOTIFICATION_ATTEMPTS + sleeps - assert worst == 154 assert worst < 300, f"worst case {worst}s exceeds the 300s visibility timeout" + assert worst < 360, f"worst case {worst}s exceeds the 360s health-stale ceiling" class TestTaskPayloads: