From 67f1182f55ec4a3f83e870136f0858d02ad56267 Mon Sep 17 00:00:00 2001 From: Audric Ackermann Date: Mon, 5 Oct 2026 08:49:34 +1100 Subject: [PATCH 1/2] feat: run the scheduled jobs one after another from 10:00 on weekdays A [queue] in jobs.toml replaces the per-job timers of github-prs-digest, zendesk-digest, crowdin-sync, snode-list and crowdin-duplicates. One session-ops-queue.timer, Mon..Fri 10:00 Australia/Melbourne, starts an anchor unit that Wants= each ready queued job; each job's drop-in orders it After= the one before, so they run in turn and a failure does not hold up the rest. crowdin-sync moves from weekly to every weekday, snode-list and crowdin-duplicates from daily to every weekday; their max_age_hours become the weekend gap plus the queue's timeouts. session-ops-silence keeps its own hourly timer. --- README.md | 4 +- deploy/README.md | 5 ++- deploy/install.sh | 41 ++++++++++++++++----- deploy/session-ops-queue.service | 10 +++++ deploy/session-ops-queue.timer | 11 ++++++ docs/jobs/crowdin-duplicates.md | 2 +- docs/jobs/crowdin-sync.md | 2 +- docs/jobs/github-prs-digest.md | 4 +- docs/jobs/session-ops-silence.md | 2 +- docs/jobs/snode-list.md | 2 +- docs/jobs/zendesk-digest.md | 4 +- src/session_ops/jobs.toml | 24 +++++++----- src/session_ops/monitor/silence.py | 6 +-- src/session_ops/ops/registry.py | 59 +++++++++++++++++++++++++++--- src/session_ops/ops/runner.py | 13 +++++-- src/session_ops/ops/units.py | 27 +++++++++----- tests/goldens/units/dropins.txt | 40 +++++--------------- tests/monitor/test_silence.py | 6 +-- tests/ops/test_deploy.py | 15 +++++++- tests/ops/test_registry.py | 26 ++++++++++++- 20 files changed, 216 insertions(+), 87 deletions(-) create mode 100644 deploy/session-ops-queue.service create mode 100644 deploy/session-ops-queue.timer diff --git a/README.md b/README.md index 7345e3e..1d5224b 100644 --- a/README.md +++ b/README.md @@ -12,8 +12,8 @@ share for translations. One package, `session_ops`, deployed to one self-hosted | `zendesk-relay` | Drafts and sends Zendesk replies from `claude:` private notes | [zendesk-relay](docs/jobs/zendesk-relay.md) | | `github-prs-digest` | Weekday Discord digest of open pull requests from outside contributors | [github-prs-digest](docs/jobs/github-prs-digest.md) | | `crowdin-duplicates` | Crowdin string slots holding more than one translation, as they open and close | [crowdin-duplicates](docs/jobs/crowdin-duplicates.md) | -| `crowdin-sync` | Weekly: Crowdin translations into pull requests on iOS, Android and the localization module | [crowdin-sync](docs/jobs/crowdin-sync.md) | -| `snode-list` | Daily: the fallback service node list into session-ios | [snode-list](docs/jobs/snode-list.md) | +| `crowdin-sync` | Weekday: Crowdin translations into pull requests on iOS, Android and the localization module | [crowdin-sync](docs/jobs/crowdin-sync.md) | +| `snode-list` | Weekday: the fallback service node list into session-ios | [snode-list](docs/jobs/snode-list.md) | | `release-stats` | On demand: download counts of the latest releases | [release-stats](docs/jobs/release-stats.md) | | `session-ops-silence` | Discord alerts for a job that failed, or stopped running | [session-ops-silence](docs/jobs/session-ops-silence.md) | diff --git a/deploy/README.md b/deploy/README.md index ed361ac..e3414a5 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -7,6 +7,7 @@ does the work: accounts, venv, env files, units, timers, and migrating an older | Unit | What it is | | --- | --- | | `session-ops@.timer` → `.service` | One per job; a generated drop-in sets its account, env files and schedule. | +| `session-ops-queue.timer` → `.service` | Starts the jobs in `jobs.toml`'s `[queue]`, which then run one at a time in its order. | | `zendesk-relay.service` | Always on, `127.0.0.1:8080`: Zendesk's `claude:` note webhooks. | | `session-ops-alert@.service` | Every unit's `OnFailure=` backstop; see [session-ops-silence](../docs/jobs/session-ops-silence.md). | @@ -59,7 +60,7 @@ installed from [`env/`](env/), saying what goes in it. ## Checking ```bash -systemctl list-timers 'session-ops@*' +systemctl list-timers 'session-ops*' systemctl start session-ops@.service && journalctl -fu session-ops@ systemctl start session-ops-alert@test.service # posts to the alerts channel curl -sS -o /dev/null -w '%{http_code}\n' -X POST 127.0.0.1:8080/zendesk/notes \ @@ -80,7 +81,7 @@ To end it, stop the timers, then close the rehearsal pull requests: for link in /etc/systemd/system/timers.target.wants/session-ops@*.timer; do [ -L "$link" ] && systemctl disable --now "${link##*/}" done -systemctl disable --now zendesk-relay.service +systemctl disable --now session-ops-queue.timer zendesk-relay.service for repo in session-android session-ios session-localization; do gh pr list -R "session-foundation/$repo" --state open --json number,headRefName \ -q '.[] | select(.headRefName | startswith("rehearsal/")) | .number' | diff --git a/deploy/install.sh b/deploy/install.sh index 8ebf5f4..6b78646 100755 --- a/deploy/install.sh +++ b/deploy/install.sh @@ -5,7 +5,7 @@ # It creates the accounts, builds the venv, creates any missing env file empty, moves # state and env files from the layout before session-ops@ units, installs the units # and each job's drop-ins, and enables every job whose env files have content and no -# other. It never edits nginx, which certbot owns; see deploy/README.md for the route. +# other: on its own timer, or on the queue's. It never edits nginx, which certbot owns; see deploy/README.md for the route. set -eu ROOT=$(cd "$(dirname "$0")/.." && pwd) @@ -106,27 +106,50 @@ rm -f "$UNITS/crowdin-relay.service" install -m 644 "$ROOT"/deploy/*.service "$ROOT"/deploy/*.timer "$UNITS/" # Only the generated files go, so a drop-in added by hand survives. -rm -f "$UNITS"/session-ops@*.service.d/job.conf "$UNITS"/session-ops@*.timer.d/schedule.conf +rm -f "$UNITS"/session-ops@*.service.d/job.conf "$UNITS"/session-ops@*.timer.d/schedule.conf \ + "$UNITS"/session-ops-queue.timer.d/schedule.conf "$OPS" units --out "$UNITS" >/dev/null -systemctl daemon-reload READY=$("$OPS" list --ready) -# A job removed from jobs.toml, or whose env file was emptied, stops being scheduled. +QUEUED=$("$OPS" list --queued) +listed() { printf '%s\n' $2 | grep -qxF "$1"; } +# What the queue's timer starts: its ready jobs, rebuilt from scratch each install. +WANTS="$UNITS/session-ops-queue.service.wants" +rm -rf "$WANTS" +for job in $QUEUED; do + if listed "$job" "$READY"; then + install -d -m 755 "$WANTS" + ln -s "$UNITS/session-ops@.service" "$WANTS/session-ops@$job.service" + fi +done +systemctl daemon-reload + +# A job removed from jobs.toml, whose env file was emptied, or now queued, loses its timer. for link in "$UNITS"/timers.target.wants/session-ops@*.timer; do [ -L "$link" ] || continue job=${link##*/session-ops@} job=${job%.timer} - if ! printf '%s\n' $READY | grep -qxF "$job"; then + if ! listed "$job" "$READY" || listed "$job" "$QUEUED"; then systemctl disable --now "session-ops@$job.timer" >/dev/null - echo "disabled session-ops@$job.timer (no longer a ready job)" + echo "disabled session-ops@$job.timer (no longer a ready job with a timer of its own)" fi done for job in $READY; do - systemctl enable --now "session-ops@$job.timer" >/dev/null - echo "enabled session-ops@$job.timer" + if listed "$job" "$QUEUED"; then + echo "queued session-ops@$job.service" + else + systemctl enable --now "session-ops@$job.timer" >/dev/null + echo "enabled session-ops@$job.timer" + fi done +if [ -d "$WANTS" ]; then + systemctl enable --now session-ops-queue.timer >/dev/null + echo "enabled session-ops-queue.timer" +else + systemctl disable --now session-ops-queue.timer 2>/dev/null || true +fi for job in $("$OPS" list --not-ready); do - echo "not enabled: session-ops@$job.timer (its env file is empty)" + echo "not enabled: session-ops@$job (its env file is empty)" done if [ -s "$ETC/zendesk.env" ]; then systemctl enable zendesk-relay.service >/dev/null diff --git a/deploy/session-ops-queue.service b/deploy/session-ops-queue.service new file mode 100644 index 0000000..fd0e8ca --- /dev/null +++ b/deploy/session-ops-queue.service @@ -0,0 +1,10 @@ +# Its timer's run of the jobs in jobs.toml's [queue]: install.sh links each ready one +# into session-ops-queue.service.wants/, and their generated After= runs them in order. +[Unit] +Description=Start the queued session-ops jobs +Documentation=https://github.com/session-foundation/session-shared-scripts +OnFailure=session-ops-alert@%n.service + +[Service] +Type=oneshot +ExecStart=/usr/bin/true diff --git a/deploy/session-ops-queue.timer b/deploy/session-ops-queue.timer new file mode 100644 index 0000000..fd5ff9f --- /dev/null +++ b/deploy/session-ops-queue.timer @@ -0,0 +1,11 @@ +# The schedule comes from the drop-in (OnCalendar=), written from jobs.toml's [queue]. +[Unit] +Description=Schedule for the queued session-ops jobs +Documentation=https://github.com/session-foundation/session-shared-scripts + +[Timer] +Persistent=yes +RandomizedDelaySec=2min + +[Install] +WantedBy=timers.target diff --git a/docs/jobs/crowdin-duplicates.md b/docs/jobs/crowdin-duplicates.md index 657440b..58a6ff0 100644 --- a/docs/jobs/crowdin-duplicates.md +++ b/docs/jobs/crowdin-duplicates.md @@ -6,7 +6,7 @@ for a plural string, is what gets exported. A slot is (string, locale, plural ca | | | | --- | --- | -| Runs | `session-ops@crowdin-duplicates.timer`, daily 03:00 UTC | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, last | | Secrets | `/etc/session-ops/crowdin.env`: a read-only `CROWDIN_API_TOKEN` and the channel's webhook | | Dry run | `session-ops run crowdin-duplicates --dry-run -- --locales de` | | Re-run | `systemctl start session-ops@crowdin-duplicates.service` | diff --git a/docs/jobs/crowdin-sync.md b/docs/jobs/crowdin-sync.md index bd6cdd5..320676d 100644 --- a/docs/jobs/crowdin-sync.md +++ b/docs/jobs/crowdin-sync.md @@ -6,7 +6,7 @@ straight onto session-localization's `main`, the TypeScript module Desktop and Q | | | | --- | --- | -| Runs | `session-ops@crowdin-sync.timer`, Mondays 13:00 Australia/Melbourne | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `zendesk-digest` | | Secrets | `/etc/session-ops/crowdin.env`: a read-only `CROWDIN_API_TOKEN`; `/etc/session-ops/publish.env` and the GitHub App key, to publish | | Dry run | `session-ops run crowdin-sync --dry-run`: everything but the push, with each platform's diff in the journal | | Re-run | `systemctl start session-ops@crowdin-sync.service`; one platform with `session-ops run crowdin-sync -- --only ios` | diff --git a/docs/jobs/github-prs-digest.md b/docs/jobs/github-prs-digest.md index b6cb3d6..c40dae6 100644 --- a/docs/jobs/github-prs-digest.md +++ b/docs/jobs/github-prs-digest.md @@ -7,7 +7,7 @@ has a reason to already know about: | | | | --- | --- | -| Runs | `session-ops@github-prs-digest.timer`, Mon–Fri 09:30 Australia/Melbourne | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, first in the queue | | Secrets | `/etc/session-ops/github-prs.env`: `GITHUB_PRS_TOKEN` with no scopes at all, and the channel's webhook | | Dry run | `session-ops run github-prs-digest --dry-run`; `uv run github-prs-digest --dry-run` from a checkout | | Re-run | `systemctl start session-ops@github-prs-digest.service` | @@ -30,7 +30,7 @@ much is waiting. ## Weekdays, and the state file -The timer runs `Mon..Fri`, so Monday's run has to cover the weekend — hence a 72-hour +The queue runs `Mon..Fri`, so Monday's run has to cover the weekend — hence a 72-hour window rather than a daily one. That window overlaps itself by two days on every run, and [`--state`](../../src/session_ops/github_prs/digest.py) is what stops the overlap being noise: it records which PRs reached Discord and what each one's `updated_at` was at the time. diff --git a/docs/jobs/session-ops-silence.md b/docs/jobs/session-ops-silence.md index d587131..38f82d3 100644 --- a/docs/jobs/session-ops-silence.md +++ b/docs/jobs/session-ops-silence.md @@ -28,4 +28,4 @@ through a whole schedule: nothing fails, so nothing alerts. Each scheduled unit the first check that found it missing, so a job left disabled alerts once that is older than its `max_age_hours`: every scheduled job is meant to run. -A job in `jobs.toml` with a `schedule` is watched; the `session-ops@` template stamps it. +A job in `jobs.toml` with a `schedule`, or in its `[queue]`, is watched; the `session-ops@` template stamps it. diff --git a/docs/jobs/snode-list.md b/docs/jobs/snode-list.md index 3dc0f10..087bd7b 100644 --- a/docs/jobs/snode-list.md +++ b/docs/jobs/snode-list.md @@ -7,7 +7,7 @@ cannot reach the seed nodes. A change opens, or updates, a pull request from | | | | --- | --- | -| Runs | `session-ops@snode-list.timer`, daily 13:00 Australia/Melbourne; the source updates at 10:00 UTC | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `crowdin-sync`; the source updates at 10:00 UTC | | Secrets | `/etc/session-ops/publish.env` and the GitHub App key; `CROWDIN_DISCORD_WEBHOOK_URL` from `crowdin.env` for the summary | | Dry run | `session-ops run snode-list --dry-run` | | Re-run | `systemctl start session-ops@snode-list.service` | diff --git a/docs/jobs/zendesk-digest.md b/docs/jobs/zendesk-digest.md index 0c81a42..ef7e365 100644 --- a/docs/jobs/zendesk-digest.md +++ b/docs/jobs/zendesk-digest.md @@ -4,7 +4,7 @@ Claude reviews the Zendesk tickets awaiting a reply — `new` and `open`, no app | | | | --- | --- | -| Runs | `session-ops@zendesk-digest.timer`, Mon–Fri 10:00 Australia/Melbourne: `zendesk-resolve-reviews --apply`, then `zendesk-triage` | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `github-prs-digest`: `zendesk-resolve-reviews --apply`, then `zendesk-triage` | | Secrets | `/etc/session-ops/zendesk.env`: the Zendesk API token, the triage channel's webhook, and the Claude Code CLI login of the `zendesk` account | | Dry run | `session-ops run zendesk-digest --dry-run`; `uv run zendesk-triage --window-hours 72 --dry-run` from a checkout | | Re-run | `systemctl start session-ops@zendesk-digest.service` | @@ -162,7 +162,7 @@ If a single request ever does hit the ceiling, the JSON never closes and no `str ## Schedule -Runs **Monday to Friday at 10:00 Melbourne** over a 72h window (~70 tickets) — 00:00 UTC in winter, 23:00 UTC the previous day under AEDT. The cron this replaces had to pin UTC+10 year-round and drift an hour against local time, because GitHub cron is UTC-only; `OnCalendar=` takes a named zone, which tracks daylight saving and keeps the day-of-week local as well. The timezone belongs inside the expression; there is no `Timezone=` key in a `[Timer]` and systemd ignores one silently, so check any change with `systemd-analyze calendar`. Unlike the cron, a host that was asleep at 10:00 still gets its digest once on the next boot (`Persistent=yes`). +Runs **Monday to Friday from 10:00 Melbourne**, second in the queue, over a 72h window (~70 tickets) — 00:00 UTC in winter, 23:00 UTC the previous day under AEDT. The cron this replaces had to pin UTC+10 year-round and drift an hour against local time, because GitHub cron is UTC-only; `OnCalendar=` takes a named zone, which tracks daylight saving and keeps the day-of-week local as well. The timezone belongs inside the expression; there is no `Timezone=` key in a `[Timer]` and systemd ignores one silently, so check any change with `systemd-analyze calendar`. Unlike the cron, a host that was asleep at 10:00 still gets its digest once on the next boot (`Persistent=yes`). The window is on `updated>`, not `created>`, so a ticket the requester adds detail to days after opening it is fetched again — a created-window would never see it. 72h rather than the 24h between runs so a failed run doesn't drop a day and Monday still reaches back past the weekend. Neither the overlap nor the wider net duplicates posts, because of the dedup state above. diff --git a/src/session_ops/jobs.toml b/src/session_ops/jobs.toml index 191584a..b23b351 100644 --- a/src/session_ops/jobs.toml +++ b/src/session_ops/jobs.toml @@ -11,9 +11,18 @@ # without one reports to ALERT_DISCORD_WEBHOOK_URL # unit further [Service] lines, over the template's hardening # +# max_age_hours for a queued job, its queue's: the weekend's 72 h (73 h across +# April's DST change), plus every job ahead of it running to its timeout +# # Timezones go in the schedule itself: systemd has no Timezone= key and ignores one. # Check a schedule with `systemd-analyze calendar ""`. +# One timer starts these together and each runs once the one before it has ended, +# whether or not it succeeded, so no two of them share the host or their APIs. +[queue] +schedule = "Mon..Fri 10:00 Australia/Melbourne" +jobs = ["github-prs-digest", "zendesk-digest", "crowdin-sync", "snode-list", "crowdin-duplicates"] + [[job]] name = "github-prs-digest" description = "Daily digest of contributor pull requests" @@ -23,9 +32,8 @@ user = "ghdigest" env_files = ["/etc/session-ops/github-prs.env"] env = ["GITHUB_PRS_TOKEN", "GITHUB_PRS_DISCORD_WEBHOOK_URL"] channel_env = "GITHUB_PRS_DISCORD_WEBHOOK_URL" -# Weekdays, half an hour before the Zendesk digest. The window spans the weekend; a longer -# gap (April's 73 h DST weekend, a missed run) reaches back through covered_until in --state. -schedule = "Mon..Fri 09:30 Australia/Melbourne" +# The window spans the weekend; a longer gap (April's 73 h DST weekend, a missed run) +# reaches back through covered_until in --state. max_age_hours = 80 timeout = "15min" @@ -38,7 +46,6 @@ user = "zendesk" env_files = ["/etc/session-ops/zendesk.env"] env = ["ZENDESK_SUBDOMAIN", "ZENDESK_EMAIL", "ZENDESK_API_TOKEN", "ZENDESK_DISCORD_WEBHOOK_URL"] channel_env = "ZENDESK_DISCORD_WEBHOOK_URL" -schedule = "Mon..Fri 10:00 Australia/Melbourne" max_age_hours = 80 # A full 1000-ticket resolve is ten bulk jobs polled for up to 300 s each. timeout = "90min" @@ -61,8 +68,7 @@ user = "crowdin" env_files = ["/etc/session-ops/crowdin.env"] env = ["CROWDIN_API_TOKEN", "CROWDIN_DISCORD_WEBHOOK_URL"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "*-*-* 03:00 UTC" -max_age_hours = 28 +max_age_hours = 80 # About 9 minutes; a full scan without --croql is ~110k requests, about 85 minutes. timeout = "2h" @@ -87,8 +93,7 @@ user = "publisher" env_files = ["/etc/session-ops/crowdin.env", "/etc/session-ops/publish.env"] env = ["CROWDIN_API_TOKEN", "PUBLISH_GIT_AUTHOR"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "Mon 13:00 Australia/Melbourne" -max_age_hours = 176 +max_age_hours = 80 timeout = "1h" # Readable by this unit alone, at $CREDENTIALS_DIRECTORY; publishing needs it. unit = ["LoadCredential=github-app.pem:/etc/session-ops/github-app.pem"] @@ -101,8 +106,7 @@ user = "publisher" env_files = ["/etc/session-ops/crowdin.env", "/etc/session-ops/publish.env"] env = ["PUBLISH_GIT_AUTHOR", "CROWDIN_DISCORD_WEBHOOK_URL"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "*-*-* 13:00 Australia/Melbourne" -max_age_hours = 28 +max_age_hours = 80 timeout = "10min" unit = ["LoadCredential=github-app.pem:/etc/session-ops/github-app.pem"] diff --git a/src/session_ops/monitor/silence.py b/src/session_ops/monitor/silence.py index 3e4e2d3..26e8076 100644 --- a/src/session_ops/monitor/silence.py +++ b/src/session_ops/monitor/silence.py @@ -43,8 +43,8 @@ def load_jobs(path): """The scheduled jobs in the registry: nothing is late that has no schedule.""" - return [{"name": job.name, "max_age_hours": job.max_age_hours} - for job in registry.load(path) if job.schedule] + return [{"name": job.name, "max_age_hours": job.max_age_hours, "timer": job.timer} + for job in registry.load(path) if job.scheduled] def stamp_time(stamps_dir, name): @@ -121,7 +121,7 @@ def build_message(due, host, now): else f"no success recorded since {utc(since)}") lines.append(f"• **{name}**: silent for **{duration(now - since)}** " f"(allowed {job['max_age_hours']}h), {seen}.") - lines.append(f" `systemctl list-timers session-ops@{name}.timer` · " + lines.append(f" `systemctl list-timers {job['timer']}` · " f"`journalctl -u session-ops@{name}.service -n 50 --no-pager`") return "\n".join(lines) diff --git a/src/session_ops/ops/registry.py b/src/session_ops/ops/registry.py index 1c7e4dd..4de3c46 100644 --- a/src/session_ops/ops/registry.py +++ b/src/session_ops/ops/registry.py @@ -5,11 +5,12 @@ """ import os import tomllib -from dataclasses import dataclass, field +from dataclasses import dataclass, field, replace REGISTRY = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "jobs.toml") STATE_ROOT = "/var/lib/session-ops" +QUEUE_TIMER = "session-ops-queue.timer" @dataclass(frozen=True) @@ -27,6 +28,16 @@ class Job: channel_env: str = "ALERT_DISCORD_WEBHOOK_URL" timeout: str = "30min" unit: tuple = field(default=()) + queued: bool = False + after: str = None + + @property + def scheduled(self): + return bool(self.schedule or self.queued) + + @property + def timer(self): + return QUEUE_TIMER if self.queued else f"session-ops@{self.name}.timer" @property def state_dir(self): @@ -39,9 +50,26 @@ def argv(self, dry_run=False, state_dir=None): return args + list(self.dry_run_args) if dry_run else args -def load(path=REGISTRY): +@dataclass(frozen=True) +class Queue: + """Jobs one timer starts together, each running once the one before it has ended.""" + schedule: str = None + jobs: tuple = () + + +def _read(path): with open(path, "rb") as handle: - rows = tomllib.load(handle).get("job", []) + return tomllib.load(handle) + + +def load_queue(path=REGISTRY): + row = _read(path).get("queue", {}) + return Queue(row.get("schedule"), tuple(row.get("jobs", ()))) + + +def load(path=REGISTRY): + data = _read(path) + rows = data.get("job", []) jobs, seen = [], set() for row in rows: name = row.get("name") @@ -51,14 +79,35 @@ def load(path=REGISTRY): raise ValueError(f"{path}: job {name or '?'} lacks {', '.join(missing)}") if name in seen: raise ValueError(f"{path}: job {name} is listed twice") - if row.get("schedule") and not isinstance(row.get("max_age_hours"), (int, float)): - raise ValueError(f"{path}: scheduled job {name} needs a numeric max_age_hours") seen.add(name) jobs.append(Job(**{k: tuple(v) if isinstance(v, list) else v for k, v in row.items()})) + jobs = _apply_queue(path, jobs, data.get("queue")) + for job in jobs: + if job.scheduled and not isinstance(job.max_age_hours, (int, float)): + raise ValueError(f"{path}: scheduled job {job.name} needs a numeric max_age_hours") return jobs +def _apply_queue(path, jobs, row): + if row is None: + return jobs + order = row.get("jobs") or [] + if not row.get("schedule") or not order: + raise ValueError(f"{path}: [queue] needs a schedule and jobs") + by_name = {job.name: job for job in jobs} + for name in order: + if name not in by_name: + raise ValueError(f"{path}: [queue] lists {name}, which is not a job") + if by_name[name].schedule: + raise ValueError(f"{path}: {name} is queued, so it takes no schedule of its own") + if len(set(order)) != len(order): + raise ValueError(f"{path}: [queue] lists a job twice") + previous = dict(zip(order[1:], order)) + return [replace(job, queued=True, after=previous.get(job.name)) if job.name in order else job + for job in jobs] + + def get(name, path=REGISTRY): for job in load(path): if job.name == name: diff --git a/src/session_ops/ops/runner.py b/src/session_ops/ops/runner.py index adaf20b..f4ab382 100644 --- a/src/session_ops/ops/runner.py +++ b/src/session_ops/ops/runner.py @@ -233,6 +233,8 @@ def main(argv=None): help="Only the names of scheduled jobs whose env files have content.") readiness.add_argument("--not-ready", action="store_true", help="Only the names of scheduled jobs with an empty env file.") + readiness.add_argument("--queued", action="store_true", + help="Only the names of the queued jobs, in the order they run.") run_parser = sub.add_parser("run", help="Run a job as its timer does.") run_parser.add_argument("job") run_parser.add_argument("--dry-run", action="store_true", @@ -244,16 +246,21 @@ def main(argv=None): args = parser.parse_args(argv[:argv.index("--")] if "--" in argv else argv) if args.command == "list": + queue = registry.load_queue() + if args.queued: + print("\n".join(queue.jobs)) + return for job in registry.load(): if args.ready or args.not_ready: - if job.schedule and ready(job) == args.ready: + if job.scheduled and ready(job) == args.ready: print(job.name) else: - print(f"{job.name:24} {job.schedule or 'on demand':38} {job.user}") + when = f"queue: {queue.schedule}" if job.queued else job.schedule or "on demand" + print(f"{job.name:24} {when:44} {job.user}") return if args.command == "units": from session_ops.ops import units - for path in units.write(registry.load(), args.out): + for path in units.write(registry.load(), registry.load_queue(), args.out): print(path) return try: diff --git a/src/session_ops/ops/units.py b/src/session_ops/ops/units.py index 43cd795..a0aa0b0 100644 --- a/src/session_ops/ops/units.py +++ b/src/session_ops/ops/units.py @@ -10,31 +10,40 @@ def service_dropin(job): - lines = [HEADER, "[Unit]", f"Description={job.description}", "", "[Service]", - f"User={job.user}", f"Group={job.user}"] + lines = [HEADER, "[Unit]", f"Description={job.description}"] + if job.after: + # Only orders the two when both are started, as the queue does; never pulls one in. + lines.append(f"After=session-ops@{job.after}.service") + lines += ["", "[Service]", f"User={job.user}", f"Group={job.user}"] lines += [f"EnvironmentFile={path}" for path in job.env_files] lines.append(f"TimeoutStartSec={job.timeout}") lines += list(job.unit) return "\n".join(lines) + "\n" -def timer_dropin(job): - return "\n".join([HEADER, "[Timer]", f"OnCalendar={job.schedule}"]) + "\n" +def timer_dropin(schedule): + return "\n".join([HEADER, "[Timer]", f"OnCalendar={schedule}"]) + "\n" -def dropins(jobs): - """{relative path: content} for every job.""" +def dropins(jobs, queue): + """{relative path: content} for every job, and the queue's schedule. + + Which queued jobs the queue starts is install.sh's to decide, from their env files: + it links each ready one into session-ops-queue.service.wants/. + """ files = {} + if queue.schedule: + files["session-ops-queue.timer.d/schedule.conf"] = timer_dropin(queue.schedule) for job in jobs: files[f"session-ops@{job.name}.service.d/job.conf"] = service_dropin(job) if job.schedule: - files[f"session-ops@{job.name}.timer.d/schedule.conf"] = timer_dropin(job) + files[f"session-ops@{job.name}.timer.d/schedule.conf"] = timer_dropin(job.schedule) return files -def write(jobs, out): +def write(jobs, queue, out): written = [] - for relative, content in sorted(dropins(jobs).items()): + for relative, content in sorted(dropins(jobs, queue).items()): path = os.path.join(out, relative) os.makedirs(os.path.dirname(path), exist_ok=True) with open(path, "w", encoding="utf-8") as handle: diff --git a/tests/goldens/units/dropins.txt b/tests/goldens/units/dropins.txt index 9c001ef..2f82251 100644 --- a/tests/goldens/units/dropins.txt +++ b/tests/goldens/units/dropins.txt @@ -1,8 +1,15 @@ +==> session-ops-queue.timer.d/schedule.conf <== +# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. + +[Timer] +OnCalendar=Mon..Fri 10:00 Australia/Melbourne + ==> session-ops@crowdin-duplicates.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Reconcile Crowdin duplicate translations and post what changed +After=session-ops@snode-list.service [Service] User=crowdin @@ -10,17 +17,12 @@ Group=crowdin EnvironmentFile=/etc/session-ops/crowdin.env TimeoutStartSec=2h -==> session-ops@crowdin-duplicates.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=*-*-* 03:00 UTC - ==> session-ops@crowdin-sync.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Crowdin translations into the platform repos +After=session-ops@zendesk-digest.service [Service] User=publisher @@ -30,12 +32,6 @@ EnvironmentFile=/etc/session-ops/publish.env TimeoutStartSec=1h LoadCredential=github-app.pem:/etc/session-ops/github-app.pem -==> session-ops@crowdin-sync.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon 13:00 Australia/Melbourne - ==> session-ops@github-prs-digest.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. @@ -48,12 +44,6 @@ Group=ghdigest EnvironmentFile=/etc/session-ops/github-prs.env TimeoutStartSec=15min -==> session-ops@github-prs-digest.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon..Fri 09:30 Australia/Melbourne - ==> session-ops@release-stats.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. @@ -89,6 +79,7 @@ OnCalendar=hourly [Unit] Description=Fallback service node list into session-ios +After=session-ops@crowdin-sync.service [Service] User=publisher @@ -98,17 +89,12 @@ EnvironmentFile=/etc/session-ops/publish.env TimeoutStartSec=10min LoadCredential=github-app.pem:/etc/session-ops/github-app.pem -==> session-ops@snode-list.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=*-*-* 13:00 Australia/Melbourne - ==> session-ops@zendesk-digest.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Zendesk positive-review resolver, then the triage digest +After=session-ops@github-prs-digest.service [Service] User=zendesk @@ -120,9 +106,3 @@ BindPaths=/home/zendesk ReadWritePaths=/home/zendesk RestrictAddressFamilies=AF_INET AF_INET6 AF_UNIX -==> session-ops@zendesk-digest.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon..Fri 10:00 Australia/Melbourne - diff --git a/tests/monitor/test_silence.py b/tests/monitor/test_silence.py index d8e7fb4..034edaf 100644 --- a/tests/monitor/test_silence.py +++ b/tests/monitor/test_silence.py @@ -9,7 +9,7 @@ HOUR = 3600 NOW = 1_790_000_000.0 -JOB = {"name": "github-prs-digest", "max_age_hours": 80} +JOB = {"name": "github-prs-digest", "max_age_hours": 80, "timer": "session-ops-queue.timer"} class TestEvaluate(unittest.TestCase): @@ -73,7 +73,7 @@ def test_names_the_job_its_silence_and_where_to_look(self): self.assertIn("`box`", message) self.assertIn("**github-prs-digest**: silent for **4d 4h** (allowed 80h)", message) self.assertIn("last success 2026-09-", message) - self.assertIn("systemctl list-timers session-ops@github-prs-digest.timer", message) + self.assertIn("systemctl list-timers session-ops-queue.timer", message) self.assertIn("journalctl -u session-ops@github-prs-digest.service", message) def test_a_job_that_never_succeeded_says_since_when_it_was_watched(self): @@ -85,7 +85,7 @@ class TestRegistry(unittest.TestCase): def test_every_scheduled_job_is_watched_and_nothing_else(self): from session_ops.ops import registry watched = {job["name"] for job in silence.load_jobs(silence.REGISTRY)} - self.assertEqual(watched, {job.name for job in registry.load() if job.schedule}) + self.assertEqual(watched, {job.name for job in registry.load() if job.scheduled}) if __name__ == "__main__": diff --git a/tests/ops/test_deploy.py b/tests/ops/test_deploy.py index b09214d..96778f5 100644 --- a/tests/ops/test_deploy.py +++ b/tests/ops/test_deploy.py @@ -56,14 +56,25 @@ def test_every_unit_reports_its_failure(self): self.assertIn("OnFailure=session-ops-alert@%n.service", unit_text(name)) def test_a_scheduled_job_gets_a_timer_and_an_unscheduled_one_does_not(self): - files = units.dropins(registry.load()) + files = units.dropins(registry.load(), registry.load_queue()) for job in registry.load(): with self.subTest(job=job.name): self.assertEqual(f"session-ops@{job.name}.timer.d/schedule.conf" in files, bool(job.schedule)) + def test_each_queued_job_runs_after_the_one_before_it(self): + queue = registry.load_queue() + files = units.dropins(registry.load(), queue) + self.assertIn(f"OnCalendar={queue.schedule}\n", + files["session-ops-queue.timer.d/schedule.conf"]) + for before, job in zip(queue.jobs, queue.jobs[1:]): + with self.subTest(job=job): + self.assertIn(f"\nAfter=session-ops@{before}.service\n", + files[f"session-ops@{job}.service.d/job.conf"]) + self.assertNotIn("After=", files[f"session-ops@{queue.jobs[0]}.service.d/job.conf"]) + def test_the_generated_dropins(self): - files = units.dropins(registry.load()) + files = units.dropins(registry.load(), registry.load_queue()) text = "".join(f"==> {path} <==\n{files[path]}\n" for path in sorted(files)) assert_golden(self, "units/dropins.txt", text) diff --git a/tests/ops/test_registry.py b/tests/ops/test_registry.py index 4dfc71d..9a068f5 100644 --- a/tests/ops/test_registry.py +++ b/tests/ops/test_registry.py @@ -15,6 +15,12 @@ user = "u" env_files = ["/etc/a.env"] args = ["--state", "{state}/s.json"] +max_age_hours = 80 +''' +QUEUE = ''' +[queue] +schedule = "Mon..Fri 10:00 Australia/Melbourne" +jobs = ["a", "b"] ''' @@ -40,7 +46,25 @@ def test_a_missing_field_is_refused(self): def test_a_scheduled_job_needs_a_max_age(self): with self.assertRaises(ValueError): - load(VALID + 'schedule = "daily"\n') + load(VALID.replace("max_age_hours = 80\n", "") + 'schedule = "daily"\n') + + def test_a_queued_job_runs_after_the_one_listed_before_it(self): + jobs = load(VALID + VALID.replace('"a"', '"b"') + QUEUE) + self.assertEqual([(job.queued, job.after) for job in jobs], [(True, None), (True, "a")]) + self.assertTrue(all(job.scheduled for job in jobs)) + self.assertEqual(jobs[1].timer, "session-ops-queue.timer") + + def test_a_queued_job_needs_a_max_age(self): + with self.assertRaises(ValueError): + load(VALID.replace("max_age_hours = 80\n", "") + QUEUE.replace(', "b"', "")) + + def test_a_queued_job_with_its_own_schedule_is_refused(self): + with self.assertRaises(ValueError): + load(VALID + 'schedule = "daily"\n' + QUEUE.replace(', "b"', "")) + + def test_a_queue_naming_an_unknown_job_is_refused(self): + with self.assertRaises(ValueError): + load(VALID + QUEUE) def test_a_name_listed_twice_is_refused(self): with self.assertRaises(ValueError): From 266d85e3e37dfa6542e5e86a5dcf98ecd93cdbd6 Mon Sep 17 00:00:00 2001 From: Audric Ackermann Date: Mon, 5 Oct 2026 09:28:36 +1100 Subject: [PATCH 2/2] fix: order each queued job after every job ahead of it install.sh links only the ready queued jobs into the queue, and After= orders nothing against a unit that is not being started. With only the previous job named, an empty env file in the middle of the queue let the job after it run alongside the ones before. Naming every job ahead keeps them one at a time whichever subset is ready. Also drops the last weekly/daily wording for crowdin-sync and crowdin-duplicates, and says in the deploy README to re-run one job by its own service, since starting the queue again re-runs the jobs already done. --- deploy/README.md | 2 ++ docs/jobs/crowdin-duplicates.md | 8 ++++---- src/session_ops/crowdin/generate_language_list.py | 2 +- src/session_ops/crowdin/sync.py | 2 +- src/session_ops/ops/registry.py | 8 +++++--- src/session_ops/ops/units.py | 4 ++-- tests/goldens/units/dropins.txt | 6 +++--- tests/ops/test_deploy.py | 7 ++++--- tests/ops/test_registry.py | 2 +- 9 files changed, 23 insertions(+), 18 deletions(-) diff --git a/deploy/README.md b/deploy/README.md index e3414a5..e0fe969 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -62,6 +62,8 @@ installed from [`env/`](env/), saying what goes in it. ```bash systemctl list-timers 'session-ops*' systemctl start session-ops@.service && journalctl -fu session-ops@ +# Re-run one job by its own service: starting session-ops-queue.service again re-runs +# every queued job that has already finished today. systemctl start session-ops-alert@test.service # posts to the alerts channel curl -sS -o /dev/null -w '%{http_code}\n' -X POST 127.0.0.1:8080/zendesk/notes \ -H 'Content-Type: application/json' -d '{"ticket_id":"1"}' # expect 401 diff --git a/docs/jobs/crowdin-duplicates.md b/docs/jobs/crowdin-duplicates.md index 58a6ff0..35b78f8 100644 --- a/docs/jobs/crowdin-duplicates.md +++ b/docs/jobs/crowdin-duplicates.md @@ -18,10 +18,10 @@ The open slots live in `/var/lib/session-ops/crowdin-duplicates/duplicates.json` only what changed: slots newly holding 2+ translations, and slots that no longer do. Nothing changed, nothing is posted. -Reconciliation judges every string of every locale once a day, so a new duplicate is -posted within a day, well before the weekly export. A slot whose string was deleted, -or whose locale left the project, resolves. One reconciliation runs at a time: one -started while another holds the state's lock exits. The state is written only once +Reconciliation judges every string of every locale each weekday, after that day's +export, so a new duplicate is posted before the next one. A slot whose string was +deleted, or whose locale left the project, resolves. One reconciliation runs at a time: +one started while another holds the state's lock exits. The state is written only once Discord accepted every message, so a failed post is repeated in full rather than lost. Losing the state is not harmless the way a digest's dedup file is: every open slot diff --git a/src/session_ops/crowdin/generate_language_list.py b/src/session_ops/crowdin/generate_language_list.py index fcb3770..ea6eba6 100755 --- a/src/session_ops/crowdin/generate_language_list.py +++ b/src/session_ops/crowdin/generate_language_list.py @@ -117,7 +117,7 @@ def generate_languages_ts(parsed_data: Dict[str, Any], output_path: str): keys = locale_keys(parsed_data) names, english, territories, unnamed = build_language_data(keys) - # Reported rather than raised: this runs in the weekly translation job, and a language nobody + # Reported rather than raised: this runs in the translation job, and a language nobody # has named yet must not hold up everyone else's strings. The fix is an override in # languageList.ts, which is a decision somebody makes rather than one this script can. if unnamed: diff --git a/src/session_ops/crowdin/sync.py b/src/session_ops/crowdin/sync.py index c9cb29e..f3e774d 100644 --- a/src/session_ops/crowdin/sync.py +++ b/src/session_ops/crowdin/sync.py @@ -1,5 +1,5 @@ """ -Weekly: download the approved translations from Crowdin, validate them, generate each +Weekdays: download the approved translations from Crowdin, validate them, generate each platform's strings, and publish them. - session-android and session-ios get a pull request from diff --git a/src/session_ops/ops/registry.py b/src/session_ops/ops/registry.py index 4de3c46..9ed9186 100644 --- a/src/session_ops/ops/registry.py +++ b/src/session_ops/ops/registry.py @@ -29,7 +29,7 @@ class Job: timeout: str = "30min" unit: tuple = field(default=()) queued: bool = False - after: str = None + after: tuple = () @property def scheduled(self): @@ -103,8 +103,10 @@ def _apply_queue(path, jobs, row): raise ValueError(f"{path}: {name} is queued, so it takes no schedule of its own") if len(set(order)) != len(order): raise ValueError(f"{path}: [queue] lists a job twice") - previous = dict(zip(order[1:], order)) - return [replace(job, queued=True, after=previous.get(job.name)) if job.name in order else job + # Every job ahead, not just the previous one: install.sh links only the ready ones, + # and After= orders nothing against a unit that is not being started. + ahead = {name: tuple(order[:i]) for i, name in enumerate(order)} + return [replace(job, queued=True, after=ahead[job.name]) if job.name in order else job for job in jobs] diff --git a/src/session_ops/ops/units.py b/src/session_ops/ops/units.py index a0aa0b0..0d1ec64 100644 --- a/src/session_ops/ops/units.py +++ b/src/session_ops/ops/units.py @@ -12,8 +12,8 @@ def service_dropin(job): lines = [HEADER, "[Unit]", f"Description={job.description}"] if job.after: - # Only orders the two when both are started, as the queue does; never pulls one in. - lines.append(f"After=session-ops@{job.after}.service") + # Orders this after whichever of them are started too; never pulls one in. + lines.append("After=" + " ".join(f"session-ops@{name}.service" for name in job.after)) lines += ["", "[Service]", f"User={job.user}", f"Group={job.user}"] lines += [f"EnvironmentFile={path}" for path in job.env_files] lines.append(f"TimeoutStartSec={job.timeout}") diff --git a/tests/goldens/units/dropins.txt b/tests/goldens/units/dropins.txt index 2f82251..27a9306 100644 --- a/tests/goldens/units/dropins.txt +++ b/tests/goldens/units/dropins.txt @@ -9,7 +9,7 @@ OnCalendar=Mon..Fri 10:00 Australia/Melbourne [Unit] Description=Reconcile Crowdin duplicate translations and post what changed -After=session-ops@snode-list.service +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service session-ops@crowdin-sync.service session-ops@snode-list.service [Service] User=crowdin @@ -22,7 +22,7 @@ TimeoutStartSec=2h [Unit] Description=Crowdin translations into the platform repos -After=session-ops@zendesk-digest.service +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service [Service] User=publisher @@ -79,7 +79,7 @@ OnCalendar=hourly [Unit] Description=Fallback service node list into session-ios -After=session-ops@crowdin-sync.service +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service session-ops@crowdin-sync.service [Service] User=publisher diff --git a/tests/ops/test_deploy.py b/tests/ops/test_deploy.py index 96778f5..ffe4a0f 100644 --- a/tests/ops/test_deploy.py +++ b/tests/ops/test_deploy.py @@ -62,14 +62,15 @@ def test_a_scheduled_job_gets_a_timer_and_an_unscheduled_one_does_not(self): self.assertEqual(f"session-ops@{job.name}.timer.d/schedule.conf" in files, bool(job.schedule)) - def test_each_queued_job_runs_after_the_one_before_it(self): + def test_each_queued_job_runs_after_every_one_before_it(self): queue = registry.load_queue() files = units.dropins(registry.load(), queue) self.assertIn(f"OnCalendar={queue.schedule}\n", files["session-ops-queue.timer.d/schedule.conf"]) - for before, job in zip(queue.jobs, queue.jobs[1:]): + for i, job in enumerate(queue.jobs[1:], start=1): with self.subTest(job=job): - self.assertIn(f"\nAfter=session-ops@{before}.service\n", + ahead = " ".join(f"session-ops@{name}.service" for name in queue.jobs[:i]) + self.assertIn(f"\nAfter={ahead}\n", files[f"session-ops@{job}.service.d/job.conf"]) self.assertNotIn("After=", files[f"session-ops@{queue.jobs[0]}.service.d/job.conf"]) diff --git a/tests/ops/test_registry.py b/tests/ops/test_registry.py index 9a068f5..88842f3 100644 --- a/tests/ops/test_registry.py +++ b/tests/ops/test_registry.py @@ -50,7 +50,7 @@ def test_a_scheduled_job_needs_a_max_age(self): def test_a_queued_job_runs_after_the_one_listed_before_it(self): jobs = load(VALID + VALID.replace('"a"', '"b"') + QUEUE) - self.assertEqual([(job.queued, job.after) for job in jobs], [(True, None), (True, "a")]) + self.assertEqual([(job.queued, job.after) for job in jobs], [(True, ()), (True, ("a",))]) self.assertTrue(all(job.scheduled for job in jobs)) self.assertEqual(jobs[1].timer, "session-ops-queue.timer")