diff --git a/README.md b/README.md index 7345e3e..1d5224b 100644 --- a/README.md +++ b/README.md @@ -12,8 +12,8 @@ share for translations. One package, `session_ops`, deployed to one self-hosted | `zendesk-relay` | Drafts and sends Zendesk replies from `claude:` private notes | [zendesk-relay](docs/jobs/zendesk-relay.md) | | `github-prs-digest` | Weekday Discord digest of open pull requests from outside contributors | [github-prs-digest](docs/jobs/github-prs-digest.md) | | `crowdin-duplicates` | Crowdin string slots holding more than one translation, as they open and close | [crowdin-duplicates](docs/jobs/crowdin-duplicates.md) | -| `crowdin-sync` | Weekly: Crowdin translations into pull requests on iOS, Android and the localization module | [crowdin-sync](docs/jobs/crowdin-sync.md) | -| `snode-list` | Daily: the fallback service node list into session-ios | [snode-list](docs/jobs/snode-list.md) | +| `crowdin-sync` | Weekday: Crowdin translations into pull requests on iOS, Android and the localization module | [crowdin-sync](docs/jobs/crowdin-sync.md) | +| `snode-list` | Weekday: the fallback service node list into session-ios | [snode-list](docs/jobs/snode-list.md) | | `release-stats` | On demand: download counts of the latest releases | [release-stats](docs/jobs/release-stats.md) | | `session-ops-silence` | Discord alerts for a job that failed, or stopped running | [session-ops-silence](docs/jobs/session-ops-silence.md) | diff --git a/deploy/README.md b/deploy/README.md index ed361ac..e0fe969 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -7,6 +7,7 @@ does the work: accounts, venv, env files, units, timers, and migrating an older | Unit | What it is | | --- | --- | | `session-ops@.timer` → `.service` | One per job; a generated drop-in sets its account, env files and schedule. | +| `session-ops-queue.timer` → `.service` | Starts the jobs in `jobs.toml`'s `[queue]`, which then run one at a time in its order. | | `zendesk-relay.service` | Always on, `127.0.0.1:8080`: Zendesk's `claude:` note webhooks. | | `session-ops-alert@.service` | Every unit's `OnFailure=` backstop; see [session-ops-silence](../docs/jobs/session-ops-silence.md). | @@ -59,8 +60,10 @@ installed from [`env/`](env/), saying what goes in it. ## Checking ```bash -systemctl list-timers 'session-ops@*' +systemctl list-timers 'session-ops*' systemctl start session-ops@.service && journalctl -fu session-ops@ +# Re-run one job by its own service: starting session-ops-queue.service again re-runs +# every queued job that has already finished today. systemctl start session-ops-alert@test.service # posts to the alerts channel curl -sS -o /dev/null -w '%{http_code}\n' -X POST 127.0.0.1:8080/zendesk/notes \ -H 'Content-Type: application/json' -d '{"ticket_id":"1"}' # expect 401 @@ -80,7 +83,7 @@ To end it, stop the timers, then close the rehearsal pull requests: for link in /etc/systemd/system/timers.target.wants/session-ops@*.timer; do [ -L "$link" ] && systemctl disable --now "${link##*/}" done -systemctl disable --now zendesk-relay.service +systemctl disable --now session-ops-queue.timer zendesk-relay.service for repo in session-android session-ios session-localization; do gh pr list -R "session-foundation/$repo" --state open --json number,headRefName \ -q '.[] | select(.headRefName | startswith("rehearsal/")) | .number' | diff --git a/deploy/install.sh b/deploy/install.sh index 8ebf5f4..6b78646 100755 --- a/deploy/install.sh +++ b/deploy/install.sh @@ -5,7 +5,7 @@ # It creates the accounts, builds the venv, creates any missing env file empty, moves # state and env files from the layout before session-ops@ units, installs the units # and each job's drop-ins, and enables every job whose env files have content and no -# other. It never edits nginx, which certbot owns; see deploy/README.md for the route. +# other: on its own timer, or on the queue's. It never edits nginx, which certbot owns; see deploy/README.md for the route. set -eu ROOT=$(cd "$(dirname "$0")/.." && pwd) @@ -106,27 +106,50 @@ rm -f "$UNITS/crowdin-relay.service" install -m 644 "$ROOT"/deploy/*.service "$ROOT"/deploy/*.timer "$UNITS/" # Only the generated files go, so a drop-in added by hand survives. -rm -f "$UNITS"/session-ops@*.service.d/job.conf "$UNITS"/session-ops@*.timer.d/schedule.conf +rm -f "$UNITS"/session-ops@*.service.d/job.conf "$UNITS"/session-ops@*.timer.d/schedule.conf \ + "$UNITS"/session-ops-queue.timer.d/schedule.conf "$OPS" units --out "$UNITS" >/dev/null -systemctl daemon-reload READY=$("$OPS" list --ready) -# A job removed from jobs.toml, or whose env file was emptied, stops being scheduled. +QUEUED=$("$OPS" list --queued) +listed() { printf '%s\n' $2 | grep -qxF "$1"; } +# What the queue's timer starts: its ready jobs, rebuilt from scratch each install. +WANTS="$UNITS/session-ops-queue.service.wants" +rm -rf "$WANTS" +for job in $QUEUED; do + if listed "$job" "$READY"; then + install -d -m 755 "$WANTS" + ln -s "$UNITS/session-ops@.service" "$WANTS/session-ops@$job.service" + fi +done +systemctl daemon-reload + +# A job removed from jobs.toml, whose env file was emptied, or now queued, loses its timer. for link in "$UNITS"/timers.target.wants/session-ops@*.timer; do [ -L "$link" ] || continue job=${link##*/session-ops@} job=${job%.timer} - if ! printf '%s\n' $READY | grep -qxF "$job"; then + if ! listed "$job" "$READY" || listed "$job" "$QUEUED"; then systemctl disable --now "session-ops@$job.timer" >/dev/null - echo "disabled session-ops@$job.timer (no longer a ready job)" + echo "disabled session-ops@$job.timer (no longer a ready job with a timer of its own)" fi done for job in $READY; do - systemctl enable --now "session-ops@$job.timer" >/dev/null - echo "enabled session-ops@$job.timer" + if listed "$job" "$QUEUED"; then + echo "queued session-ops@$job.service" + else + systemctl enable --now "session-ops@$job.timer" >/dev/null + echo "enabled session-ops@$job.timer" + fi done +if [ -d "$WANTS" ]; then + systemctl enable --now session-ops-queue.timer >/dev/null + echo "enabled session-ops-queue.timer" +else + systemctl disable --now session-ops-queue.timer 2>/dev/null || true +fi for job in $("$OPS" list --not-ready); do - echo "not enabled: session-ops@$job.timer (its env file is empty)" + echo "not enabled: session-ops@$job (its env file is empty)" done if [ -s "$ETC/zendesk.env" ]; then systemctl enable zendesk-relay.service >/dev/null diff --git a/deploy/session-ops-queue.service b/deploy/session-ops-queue.service new file mode 100644 index 0000000..fd0e8ca --- /dev/null +++ b/deploy/session-ops-queue.service @@ -0,0 +1,10 @@ +# Its timer's run of the jobs in jobs.toml's [queue]: install.sh links each ready one +# into session-ops-queue.service.wants/, and their generated After= runs them in order. +[Unit] +Description=Start the queued session-ops jobs +Documentation=https://github.com/session-foundation/session-shared-scripts +OnFailure=session-ops-alert@%n.service + +[Service] +Type=oneshot +ExecStart=/usr/bin/true diff --git a/deploy/session-ops-queue.timer b/deploy/session-ops-queue.timer new file mode 100644 index 0000000..fd5ff9f --- /dev/null +++ b/deploy/session-ops-queue.timer @@ -0,0 +1,11 @@ +# The schedule comes from the drop-in (OnCalendar=), written from jobs.toml's [queue]. +[Unit] +Description=Schedule for the queued session-ops jobs +Documentation=https://github.com/session-foundation/session-shared-scripts + +[Timer] +Persistent=yes +RandomizedDelaySec=2min + +[Install] +WantedBy=timers.target diff --git a/docs/jobs/crowdin-duplicates.md b/docs/jobs/crowdin-duplicates.md index 657440b..35b78f8 100644 --- a/docs/jobs/crowdin-duplicates.md +++ b/docs/jobs/crowdin-duplicates.md @@ -6,7 +6,7 @@ for a plural string, is what gets exported. A slot is (string, locale, plural ca | | | | --- | --- | -| Runs | `session-ops@crowdin-duplicates.timer`, daily 03:00 UTC | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, last | | Secrets | `/etc/session-ops/crowdin.env`: a read-only `CROWDIN_API_TOKEN` and the channel's webhook | | Dry run | `session-ops run crowdin-duplicates --dry-run -- --locales de` | | Re-run | `systemctl start session-ops@crowdin-duplicates.service` | @@ -18,10 +18,10 @@ The open slots live in `/var/lib/session-ops/crowdin-duplicates/duplicates.json` only what changed: slots newly holding 2+ translations, and slots that no longer do. Nothing changed, nothing is posted. -Reconciliation judges every string of every locale once a day, so a new duplicate is -posted within a day, well before the weekly export. A slot whose string was deleted, -or whose locale left the project, resolves. One reconciliation runs at a time: one -started while another holds the state's lock exits. The state is written only once +Reconciliation judges every string of every locale each weekday, after that day's +export, so a new duplicate is posted before the next one. A slot whose string was +deleted, or whose locale left the project, resolves. One reconciliation runs at a time: +one started while another holds the state's lock exits. The state is written only once Discord accepted every message, so a failed post is repeated in full rather than lost. Losing the state is not harmless the way a digest's dedup file is: every open slot diff --git a/docs/jobs/crowdin-sync.md b/docs/jobs/crowdin-sync.md index bd6cdd5..320676d 100644 --- a/docs/jobs/crowdin-sync.md +++ b/docs/jobs/crowdin-sync.md @@ -6,7 +6,7 @@ straight onto session-localization's `main`, the TypeScript module Desktop and Q | | | | --- | --- | -| Runs | `session-ops@crowdin-sync.timer`, Mondays 13:00 Australia/Melbourne | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `zendesk-digest` | | Secrets | `/etc/session-ops/crowdin.env`: a read-only `CROWDIN_API_TOKEN`; `/etc/session-ops/publish.env` and the GitHub App key, to publish | | Dry run | `session-ops run crowdin-sync --dry-run`: everything but the push, with each platform's diff in the journal | | Re-run | `systemctl start session-ops@crowdin-sync.service`; one platform with `session-ops run crowdin-sync -- --only ios` | diff --git a/docs/jobs/github-prs-digest.md b/docs/jobs/github-prs-digest.md index b6cb3d6..c40dae6 100644 --- a/docs/jobs/github-prs-digest.md +++ b/docs/jobs/github-prs-digest.md @@ -7,7 +7,7 @@ has a reason to already know about: | | | | --- | --- | -| Runs | `session-ops@github-prs-digest.timer`, Mon–Fri 09:30 Australia/Melbourne | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, first in the queue | | Secrets | `/etc/session-ops/github-prs.env`: `GITHUB_PRS_TOKEN` with no scopes at all, and the channel's webhook | | Dry run | `session-ops run github-prs-digest --dry-run`; `uv run github-prs-digest --dry-run` from a checkout | | Re-run | `systemctl start session-ops@github-prs-digest.service` | @@ -30,7 +30,7 @@ much is waiting. ## Weekdays, and the state file -The timer runs `Mon..Fri`, so Monday's run has to cover the weekend — hence a 72-hour +The queue runs `Mon..Fri`, so Monday's run has to cover the weekend — hence a 72-hour window rather than a daily one. That window overlaps itself by two days on every run, and [`--state`](../../src/session_ops/github_prs/digest.py) is what stops the overlap being noise: it records which PRs reached Discord and what each one's `updated_at` was at the time. diff --git a/docs/jobs/session-ops-silence.md b/docs/jobs/session-ops-silence.md index d587131..38f82d3 100644 --- a/docs/jobs/session-ops-silence.md +++ b/docs/jobs/session-ops-silence.md @@ -28,4 +28,4 @@ through a whole schedule: nothing fails, so nothing alerts. Each scheduled unit the first check that found it missing, so a job left disabled alerts once that is older than its `max_age_hours`: every scheduled job is meant to run. -A job in `jobs.toml` with a `schedule` is watched; the `session-ops@` template stamps it. +A job in `jobs.toml` with a `schedule`, or in its `[queue]`, is watched; the `session-ops@` template stamps it. diff --git a/docs/jobs/snode-list.md b/docs/jobs/snode-list.md index 3dc0f10..087bd7b 100644 --- a/docs/jobs/snode-list.md +++ b/docs/jobs/snode-list.md @@ -7,7 +7,7 @@ cannot reach the seed nodes. A change opens, or updates, a pull request from | | | | --- | --- | -| Runs | `session-ops@snode-list.timer`, daily 13:00 Australia/Melbourne; the source updates at 10:00 UTC | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `crowdin-sync`; the source updates at 10:00 UTC | | Secrets | `/etc/session-ops/publish.env` and the GitHub App key; `CROWDIN_DISCORD_WEBHOOK_URL` from `crowdin.env` for the summary | | Dry run | `session-ops run snode-list --dry-run` | | Re-run | `systemctl start session-ops@snode-list.service` | diff --git a/docs/jobs/zendesk-digest.md b/docs/jobs/zendesk-digest.md index 0c81a42..ef7e365 100644 --- a/docs/jobs/zendesk-digest.md +++ b/docs/jobs/zendesk-digest.md @@ -4,7 +4,7 @@ Claude reviews the Zendesk tickets awaiting a reply — `new` and `open`, no app | | | | --- | --- | -| Runs | `session-ops@zendesk-digest.timer`, Mon–Fri 10:00 Australia/Melbourne: `zendesk-resolve-reviews --apply`, then `zendesk-triage` | +| Runs | `session-ops-queue.timer`, Mon–Fri from 10:00 Australia/Melbourne, after `github-prs-digest`: `zendesk-resolve-reviews --apply`, then `zendesk-triage` | | Secrets | `/etc/session-ops/zendesk.env`: the Zendesk API token, the triage channel's webhook, and the Claude Code CLI login of the `zendesk` account | | Dry run | `session-ops run zendesk-digest --dry-run`; `uv run zendesk-triage --window-hours 72 --dry-run` from a checkout | | Re-run | `systemctl start session-ops@zendesk-digest.service` | @@ -162,7 +162,7 @@ If a single request ever does hit the ceiling, the JSON never closes and no `str ## Schedule -Runs **Monday to Friday at 10:00 Melbourne** over a 72h window (~70 tickets) — 00:00 UTC in winter, 23:00 UTC the previous day under AEDT. The cron this replaces had to pin UTC+10 year-round and drift an hour against local time, because GitHub cron is UTC-only; `OnCalendar=` takes a named zone, which tracks daylight saving and keeps the day-of-week local as well. The timezone belongs inside the expression; there is no `Timezone=` key in a `[Timer]` and systemd ignores one silently, so check any change with `systemd-analyze calendar`. Unlike the cron, a host that was asleep at 10:00 still gets its digest once on the next boot (`Persistent=yes`). +Runs **Monday to Friday from 10:00 Melbourne**, second in the queue, over a 72h window (~70 tickets) — 00:00 UTC in winter, 23:00 UTC the previous day under AEDT. The cron this replaces had to pin UTC+10 year-round and drift an hour against local time, because GitHub cron is UTC-only; `OnCalendar=` takes a named zone, which tracks daylight saving and keeps the day-of-week local as well. The timezone belongs inside the expression; there is no `Timezone=` key in a `[Timer]` and systemd ignores one silently, so check any change with `systemd-analyze calendar`. Unlike the cron, a host that was asleep at 10:00 still gets its digest once on the next boot (`Persistent=yes`). The window is on `updated>`, not `created>`, so a ticket the requester adds detail to days after opening it is fetched again — a created-window would never see it. 72h rather than the 24h between runs so a failed run doesn't drop a day and Monday still reaches back past the weekend. Neither the overlap nor the wider net duplicates posts, because of the dedup state above. diff --git a/src/session_ops/crowdin/generate_language_list.py b/src/session_ops/crowdin/generate_language_list.py index fcb3770..ea6eba6 100755 --- a/src/session_ops/crowdin/generate_language_list.py +++ b/src/session_ops/crowdin/generate_language_list.py @@ -117,7 +117,7 @@ def generate_languages_ts(parsed_data: Dict[str, Any], output_path: str): keys = locale_keys(parsed_data) names, english, territories, unnamed = build_language_data(keys) - # Reported rather than raised: this runs in the weekly translation job, and a language nobody + # Reported rather than raised: this runs in the translation job, and a language nobody # has named yet must not hold up everyone else's strings. The fix is an override in # languageList.ts, which is a decision somebody makes rather than one this script can. if unnamed: diff --git a/src/session_ops/crowdin/sync.py b/src/session_ops/crowdin/sync.py index c9cb29e..f3e774d 100644 --- a/src/session_ops/crowdin/sync.py +++ b/src/session_ops/crowdin/sync.py @@ -1,5 +1,5 @@ """ -Weekly: download the approved translations from Crowdin, validate them, generate each +Weekdays: download the approved translations from Crowdin, validate them, generate each platform's strings, and publish them. - session-android and session-ios get a pull request from diff --git a/src/session_ops/jobs.toml b/src/session_ops/jobs.toml index 191584a..b23b351 100644 --- a/src/session_ops/jobs.toml +++ b/src/session_ops/jobs.toml @@ -11,9 +11,18 @@ # without one reports to ALERT_DISCORD_WEBHOOK_URL # unit further [Service] lines, over the template's hardening # +# max_age_hours for a queued job, its queue's: the weekend's 72 h (73 h across +# April's DST change), plus every job ahead of it running to its timeout +# # Timezones go in the schedule itself: systemd has no Timezone= key and ignores one. # Check a schedule with `systemd-analyze calendar ""`. +# One timer starts these together and each runs once the one before it has ended, +# whether or not it succeeded, so no two of them share the host or their APIs. +[queue] +schedule = "Mon..Fri 10:00 Australia/Melbourne" +jobs = ["github-prs-digest", "zendesk-digest", "crowdin-sync", "snode-list", "crowdin-duplicates"] + [[job]] name = "github-prs-digest" description = "Daily digest of contributor pull requests" @@ -23,9 +32,8 @@ user = "ghdigest" env_files = ["/etc/session-ops/github-prs.env"] env = ["GITHUB_PRS_TOKEN", "GITHUB_PRS_DISCORD_WEBHOOK_URL"] channel_env = "GITHUB_PRS_DISCORD_WEBHOOK_URL" -# Weekdays, half an hour before the Zendesk digest. The window spans the weekend; a longer -# gap (April's 73 h DST weekend, a missed run) reaches back through covered_until in --state. -schedule = "Mon..Fri 09:30 Australia/Melbourne" +# The window spans the weekend; a longer gap (April's 73 h DST weekend, a missed run) +# reaches back through covered_until in --state. max_age_hours = 80 timeout = "15min" @@ -38,7 +46,6 @@ user = "zendesk" env_files = ["/etc/session-ops/zendesk.env"] env = ["ZENDESK_SUBDOMAIN", "ZENDESK_EMAIL", "ZENDESK_API_TOKEN", "ZENDESK_DISCORD_WEBHOOK_URL"] channel_env = "ZENDESK_DISCORD_WEBHOOK_URL" -schedule = "Mon..Fri 10:00 Australia/Melbourne" max_age_hours = 80 # A full 1000-ticket resolve is ten bulk jobs polled for up to 300 s each. timeout = "90min" @@ -61,8 +68,7 @@ user = "crowdin" env_files = ["/etc/session-ops/crowdin.env"] env = ["CROWDIN_API_TOKEN", "CROWDIN_DISCORD_WEBHOOK_URL"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "*-*-* 03:00 UTC" -max_age_hours = 28 +max_age_hours = 80 # About 9 minutes; a full scan without --croql is ~110k requests, about 85 minutes. timeout = "2h" @@ -87,8 +93,7 @@ user = "publisher" env_files = ["/etc/session-ops/crowdin.env", "/etc/session-ops/publish.env"] env = ["CROWDIN_API_TOKEN", "PUBLISH_GIT_AUTHOR"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "Mon 13:00 Australia/Melbourne" -max_age_hours = 176 +max_age_hours = 80 timeout = "1h" # Readable by this unit alone, at $CREDENTIALS_DIRECTORY; publishing needs it. unit = ["LoadCredential=github-app.pem:/etc/session-ops/github-app.pem"] @@ -101,8 +106,7 @@ user = "publisher" env_files = ["/etc/session-ops/crowdin.env", "/etc/session-ops/publish.env"] env = ["PUBLISH_GIT_AUTHOR", "CROWDIN_DISCORD_WEBHOOK_URL"] channel_env = "CROWDIN_DISCORD_WEBHOOK_URL" -schedule = "*-*-* 13:00 Australia/Melbourne" -max_age_hours = 28 +max_age_hours = 80 timeout = "10min" unit = ["LoadCredential=github-app.pem:/etc/session-ops/github-app.pem"] diff --git a/src/session_ops/monitor/silence.py b/src/session_ops/monitor/silence.py index 3e4e2d3..26e8076 100644 --- a/src/session_ops/monitor/silence.py +++ b/src/session_ops/monitor/silence.py @@ -43,8 +43,8 @@ def load_jobs(path): """The scheduled jobs in the registry: nothing is late that has no schedule.""" - return [{"name": job.name, "max_age_hours": job.max_age_hours} - for job in registry.load(path) if job.schedule] + return [{"name": job.name, "max_age_hours": job.max_age_hours, "timer": job.timer} + for job in registry.load(path) if job.scheduled] def stamp_time(stamps_dir, name): @@ -121,7 +121,7 @@ def build_message(due, host, now): else f"no success recorded since {utc(since)}") lines.append(f"• **{name}**: silent for **{duration(now - since)}** " f"(allowed {job['max_age_hours']}h), {seen}.") - lines.append(f" `systemctl list-timers session-ops@{name}.timer` · " + lines.append(f" `systemctl list-timers {job['timer']}` · " f"`journalctl -u session-ops@{name}.service -n 50 --no-pager`") return "\n".join(lines) diff --git a/src/session_ops/ops/registry.py b/src/session_ops/ops/registry.py index 1c7e4dd..9ed9186 100644 --- a/src/session_ops/ops/registry.py +++ b/src/session_ops/ops/registry.py @@ -5,11 +5,12 @@ """ import os import tomllib -from dataclasses import dataclass, field +from dataclasses import dataclass, field, replace REGISTRY = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "jobs.toml") STATE_ROOT = "/var/lib/session-ops" +QUEUE_TIMER = "session-ops-queue.timer" @dataclass(frozen=True) @@ -27,6 +28,16 @@ class Job: channel_env: str = "ALERT_DISCORD_WEBHOOK_URL" timeout: str = "30min" unit: tuple = field(default=()) + queued: bool = False + after: tuple = () + + @property + def scheduled(self): + return bool(self.schedule or self.queued) + + @property + def timer(self): + return QUEUE_TIMER if self.queued else f"session-ops@{self.name}.timer" @property def state_dir(self): @@ -39,9 +50,26 @@ def argv(self, dry_run=False, state_dir=None): return args + list(self.dry_run_args) if dry_run else args -def load(path=REGISTRY): +@dataclass(frozen=True) +class Queue: + """Jobs one timer starts together, each running once the one before it has ended.""" + schedule: str = None + jobs: tuple = () + + +def _read(path): with open(path, "rb") as handle: - rows = tomllib.load(handle).get("job", []) + return tomllib.load(handle) + + +def load_queue(path=REGISTRY): + row = _read(path).get("queue", {}) + return Queue(row.get("schedule"), tuple(row.get("jobs", ()))) + + +def load(path=REGISTRY): + data = _read(path) + rows = data.get("job", []) jobs, seen = [], set() for row in rows: name = row.get("name") @@ -51,14 +79,37 @@ def load(path=REGISTRY): raise ValueError(f"{path}: job {name or '?'} lacks {', '.join(missing)}") if name in seen: raise ValueError(f"{path}: job {name} is listed twice") - if row.get("schedule") and not isinstance(row.get("max_age_hours"), (int, float)): - raise ValueError(f"{path}: scheduled job {name} needs a numeric max_age_hours") seen.add(name) jobs.append(Job(**{k: tuple(v) if isinstance(v, list) else v for k, v in row.items()})) + jobs = _apply_queue(path, jobs, data.get("queue")) + for job in jobs: + if job.scheduled and not isinstance(job.max_age_hours, (int, float)): + raise ValueError(f"{path}: scheduled job {job.name} needs a numeric max_age_hours") return jobs +def _apply_queue(path, jobs, row): + if row is None: + return jobs + order = row.get("jobs") or [] + if not row.get("schedule") or not order: + raise ValueError(f"{path}: [queue] needs a schedule and jobs") + by_name = {job.name: job for job in jobs} + for name in order: + if name not in by_name: + raise ValueError(f"{path}: [queue] lists {name}, which is not a job") + if by_name[name].schedule: + raise ValueError(f"{path}: {name} is queued, so it takes no schedule of its own") + if len(set(order)) != len(order): + raise ValueError(f"{path}: [queue] lists a job twice") + # Every job ahead, not just the previous one: install.sh links only the ready ones, + # and After= orders nothing against a unit that is not being started. + ahead = {name: tuple(order[:i]) for i, name in enumerate(order)} + return [replace(job, queued=True, after=ahead[job.name]) if job.name in order else job + for job in jobs] + + def get(name, path=REGISTRY): for job in load(path): if job.name == name: diff --git a/src/session_ops/ops/runner.py b/src/session_ops/ops/runner.py index adaf20b..f4ab382 100644 --- a/src/session_ops/ops/runner.py +++ b/src/session_ops/ops/runner.py @@ -233,6 +233,8 @@ def main(argv=None): help="Only the names of scheduled jobs whose env files have content.") readiness.add_argument("--not-ready", action="store_true", help="Only the names of scheduled jobs with an empty env file.") + readiness.add_argument("--queued", action="store_true", + help="Only the names of the queued jobs, in the order they run.") run_parser = sub.add_parser("run", help="Run a job as its timer does.") run_parser.add_argument("job") run_parser.add_argument("--dry-run", action="store_true", @@ -244,16 +246,21 @@ def main(argv=None): args = parser.parse_args(argv[:argv.index("--")] if "--" in argv else argv) if args.command == "list": + queue = registry.load_queue() + if args.queued: + print("\n".join(queue.jobs)) + return for job in registry.load(): if args.ready or args.not_ready: - if job.schedule and ready(job) == args.ready: + if job.scheduled and ready(job) == args.ready: print(job.name) else: - print(f"{job.name:24} {job.schedule or 'on demand':38} {job.user}") + when = f"queue: {queue.schedule}" if job.queued else job.schedule or "on demand" + print(f"{job.name:24} {when:44} {job.user}") return if args.command == "units": from session_ops.ops import units - for path in units.write(registry.load(), args.out): + for path in units.write(registry.load(), registry.load_queue(), args.out): print(path) return try: diff --git a/src/session_ops/ops/units.py b/src/session_ops/ops/units.py index 43cd795..0d1ec64 100644 --- a/src/session_ops/ops/units.py +++ b/src/session_ops/ops/units.py @@ -10,31 +10,40 @@ def service_dropin(job): - lines = [HEADER, "[Unit]", f"Description={job.description}", "", "[Service]", - f"User={job.user}", f"Group={job.user}"] + lines = [HEADER, "[Unit]", f"Description={job.description}"] + if job.after: + # Orders this after whichever of them are started too; never pulls one in. + lines.append("After=" + " ".join(f"session-ops@{name}.service" for name in job.after)) + lines += ["", "[Service]", f"User={job.user}", f"Group={job.user}"] lines += [f"EnvironmentFile={path}" for path in job.env_files] lines.append(f"TimeoutStartSec={job.timeout}") lines += list(job.unit) return "\n".join(lines) + "\n" -def timer_dropin(job): - return "\n".join([HEADER, "[Timer]", f"OnCalendar={job.schedule}"]) + "\n" +def timer_dropin(schedule): + return "\n".join([HEADER, "[Timer]", f"OnCalendar={schedule}"]) + "\n" -def dropins(jobs): - """{relative path: content} for every job.""" +def dropins(jobs, queue): + """{relative path: content} for every job, and the queue's schedule. + + Which queued jobs the queue starts is install.sh's to decide, from their env files: + it links each ready one into session-ops-queue.service.wants/. + """ files = {} + if queue.schedule: + files["session-ops-queue.timer.d/schedule.conf"] = timer_dropin(queue.schedule) for job in jobs: files[f"session-ops@{job.name}.service.d/job.conf"] = service_dropin(job) if job.schedule: - files[f"session-ops@{job.name}.timer.d/schedule.conf"] = timer_dropin(job) + files[f"session-ops@{job.name}.timer.d/schedule.conf"] = timer_dropin(job.schedule) return files -def write(jobs, out): +def write(jobs, queue, out): written = [] - for relative, content in sorted(dropins(jobs).items()): + for relative, content in sorted(dropins(jobs, queue).items()): path = os.path.join(out, relative) os.makedirs(os.path.dirname(path), exist_ok=True) with open(path, "w", encoding="utf-8") as handle: diff --git a/tests/goldens/units/dropins.txt b/tests/goldens/units/dropins.txt index 9c001ef..27a9306 100644 --- a/tests/goldens/units/dropins.txt +++ b/tests/goldens/units/dropins.txt @@ -1,8 +1,15 @@ +==> session-ops-queue.timer.d/schedule.conf <== +# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. + +[Timer] +OnCalendar=Mon..Fri 10:00 Australia/Melbourne + ==> session-ops@crowdin-duplicates.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Reconcile Crowdin duplicate translations and post what changed +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service session-ops@crowdin-sync.service session-ops@snode-list.service [Service] User=crowdin @@ -10,17 +17,12 @@ Group=crowdin EnvironmentFile=/etc/session-ops/crowdin.env TimeoutStartSec=2h -==> session-ops@crowdin-duplicates.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=*-*-* 03:00 UTC - ==> session-ops@crowdin-sync.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Crowdin translations into the platform repos +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service [Service] User=publisher @@ -30,12 +32,6 @@ EnvironmentFile=/etc/session-ops/publish.env TimeoutStartSec=1h LoadCredential=github-app.pem:/etc/session-ops/github-app.pem -==> session-ops@crowdin-sync.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon 13:00 Australia/Melbourne - ==> session-ops@github-prs-digest.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. @@ -48,12 +44,6 @@ Group=ghdigest EnvironmentFile=/etc/session-ops/github-prs.env TimeoutStartSec=15min -==> session-ops@github-prs-digest.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon..Fri 09:30 Australia/Melbourne - ==> session-ops@release-stats.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. @@ -89,6 +79,7 @@ OnCalendar=hourly [Unit] Description=Fallback service node list into session-ios +After=session-ops@github-prs-digest.service session-ops@zendesk-digest.service session-ops@crowdin-sync.service [Service] User=publisher @@ -98,17 +89,12 @@ EnvironmentFile=/etc/session-ops/publish.env TimeoutStartSec=10min LoadCredential=github-app.pem:/etc/session-ops/github-app.pem -==> session-ops@snode-list.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=*-*-* 13:00 Australia/Melbourne - ==> session-ops@zendesk-digest.service.d/job.conf <== # Generated by `session-ops units` from jobs.toml. Edit the registry, not this. [Unit] Description=Zendesk positive-review resolver, then the triage digest +After=session-ops@github-prs-digest.service [Service] User=zendesk @@ -120,9 +106,3 @@ BindPaths=/home/zendesk ReadWritePaths=/home/zendesk RestrictAddressFamilies=AF_INET AF_INET6 AF_UNIX -==> session-ops@zendesk-digest.timer.d/schedule.conf <== -# Generated by `session-ops units` from jobs.toml. Edit the registry, not this. - -[Timer] -OnCalendar=Mon..Fri 10:00 Australia/Melbourne - diff --git a/tests/monitor/test_silence.py b/tests/monitor/test_silence.py index d8e7fb4..034edaf 100644 --- a/tests/monitor/test_silence.py +++ b/tests/monitor/test_silence.py @@ -9,7 +9,7 @@ HOUR = 3600 NOW = 1_790_000_000.0 -JOB = {"name": "github-prs-digest", "max_age_hours": 80} +JOB = {"name": "github-prs-digest", "max_age_hours": 80, "timer": "session-ops-queue.timer"} class TestEvaluate(unittest.TestCase): @@ -73,7 +73,7 @@ def test_names_the_job_its_silence_and_where_to_look(self): self.assertIn("`box`", message) self.assertIn("**github-prs-digest**: silent for **4d 4h** (allowed 80h)", message) self.assertIn("last success 2026-09-", message) - self.assertIn("systemctl list-timers session-ops@github-prs-digest.timer", message) + self.assertIn("systemctl list-timers session-ops-queue.timer", message) self.assertIn("journalctl -u session-ops@github-prs-digest.service", message) def test_a_job_that_never_succeeded_says_since_when_it_was_watched(self): @@ -85,7 +85,7 @@ class TestRegistry(unittest.TestCase): def test_every_scheduled_job_is_watched_and_nothing_else(self): from session_ops.ops import registry watched = {job["name"] for job in silence.load_jobs(silence.REGISTRY)} - self.assertEqual(watched, {job.name for job in registry.load() if job.schedule}) + self.assertEqual(watched, {job.name for job in registry.load() if job.scheduled}) if __name__ == "__main__": diff --git a/tests/ops/test_deploy.py b/tests/ops/test_deploy.py index b09214d..ffe4a0f 100644 --- a/tests/ops/test_deploy.py +++ b/tests/ops/test_deploy.py @@ -56,14 +56,26 @@ def test_every_unit_reports_its_failure(self): self.assertIn("OnFailure=session-ops-alert@%n.service", unit_text(name)) def test_a_scheduled_job_gets_a_timer_and_an_unscheduled_one_does_not(self): - files = units.dropins(registry.load()) + files = units.dropins(registry.load(), registry.load_queue()) for job in registry.load(): with self.subTest(job=job.name): self.assertEqual(f"session-ops@{job.name}.timer.d/schedule.conf" in files, bool(job.schedule)) + def test_each_queued_job_runs_after_every_one_before_it(self): + queue = registry.load_queue() + files = units.dropins(registry.load(), queue) + self.assertIn(f"OnCalendar={queue.schedule}\n", + files["session-ops-queue.timer.d/schedule.conf"]) + for i, job in enumerate(queue.jobs[1:], start=1): + with self.subTest(job=job): + ahead = " ".join(f"session-ops@{name}.service" for name in queue.jobs[:i]) + self.assertIn(f"\nAfter={ahead}\n", + files[f"session-ops@{job}.service.d/job.conf"]) + self.assertNotIn("After=", files[f"session-ops@{queue.jobs[0]}.service.d/job.conf"]) + def test_the_generated_dropins(self): - files = units.dropins(registry.load()) + files = units.dropins(registry.load(), registry.load_queue()) text = "".join(f"==> {path} <==\n{files[path]}\n" for path in sorted(files)) assert_golden(self, "units/dropins.txt", text) diff --git a/tests/ops/test_registry.py b/tests/ops/test_registry.py index 4dfc71d..88842f3 100644 --- a/tests/ops/test_registry.py +++ b/tests/ops/test_registry.py @@ -15,6 +15,12 @@ user = "u" env_files = ["/etc/a.env"] args = ["--state", "{state}/s.json"] +max_age_hours = 80 +''' +QUEUE = ''' +[queue] +schedule = "Mon..Fri 10:00 Australia/Melbourne" +jobs = ["a", "b"] ''' @@ -40,7 +46,25 @@ def test_a_missing_field_is_refused(self): def test_a_scheduled_job_needs_a_max_age(self): with self.assertRaises(ValueError): - load(VALID + 'schedule = "daily"\n') + load(VALID.replace("max_age_hours = 80\n", "") + 'schedule = "daily"\n') + + def test_a_queued_job_runs_after_the_one_listed_before_it(self): + jobs = load(VALID + VALID.replace('"a"', '"b"') + QUEUE) + self.assertEqual([(job.queued, job.after) for job in jobs], [(True, ()), (True, ("a",))]) + self.assertTrue(all(job.scheduled for job in jobs)) + self.assertEqual(jobs[1].timer, "session-ops-queue.timer") + + def test_a_queued_job_needs_a_max_age(self): + with self.assertRaises(ValueError): + load(VALID.replace("max_age_hours = 80\n", "") + QUEUE.replace(', "b"', "")) + + def test_a_queued_job_with_its_own_schedule_is_refused(self): + with self.assertRaises(ValueError): + load(VALID + 'schedule = "daily"\n' + QUEUE.replace(', "b"', "")) + + def test_a_queue_naming_an_unknown_job_is_refused(self): + with self.assertRaises(ValueError): + load(VALID + QUEUE) def test_a_name_listed_twice_is_refused(self): with self.assertRaises(ValueError):