From 7e36f42ad99c46c75d23e63c91880180e94c1efa Mon Sep 17 00:00:00 2001 From: Audric Ackermann Date: Tue, 6 Oct 2026 15:29:16 +1100 Subject: [PATCH] feat: list at most 10 duplicate slots per locale in Discord A locale past 10 new or resolved slots lists the first 9, then the count left and a link to the locale in the Crowdin editor, rather than splitting into "(cont.)" embeds. A one-locale burst like Korean's 462 slots is one message instead of 14, which ran into Discord's per-channel rate limit and failed the run. --- docs/jobs/crowdin-duplicates.md | 2 ++ src/session_ops/crowdin/duplicates.py | 31 +++++++++++++------------ tests/crowdin/test_duplicates.py | 33 +++++++++++++++++++++------ 3 files changed, 45 insertions(+), 21 deletions(-) diff --git a/docs/jobs/crowdin-duplicates.md b/docs/jobs/crowdin-duplicates.md index 35b78f8..317838f 100644 --- a/docs/jobs/crowdin-duplicates.md +++ b/docs/jobs/crowdin-duplicates.md @@ -17,6 +17,8 @@ for a plural string, is what gets exported. A slot is (string, locale, plural ca The open slots live in `/var/lib/session-ops/crowdin-duplicates/duplicates.json`. Each run posts only what changed: slots newly holding 2+ translations, and slots that no longer do. Nothing changed, nothing is posted. +Each locale is one embed listing up to 10 slots with editor links; past that, it gives +the count and a link to the locale in the editor, so a bulk import stays one message. Reconciliation judges every string of every locale each weekday, after that day's export, so a new duplicate is posted before the next one. A slot whose string was diff --git a/src/session_ops/crowdin/duplicates.py b/src/session_ops/crowdin/duplicates.py index e812353..e35fc51 100644 --- a/src/session_ops/crowdin/duplicates.py +++ b/src/session_ops/crowdin/duplicates.py @@ -19,6 +19,10 @@ STATE_VERSION = 1 OPEN_COLOR, RESOLVED_COLOR = 0xE67E22, 0x2ECC71 +# A locale's slots listed one by one; past this, a count and the locale's editor. A burst +# (a bulk import, a whole locale reviewed at once) then stays a message, not a flood that +# outruns Discord's per-channel rate limit. +LISTED_PER_LOCALE = 10 def user_label(u): @@ -79,9 +83,12 @@ def __init__(self, details): for lang in details["targetLanguages"]} self.locales = details["targetLanguageIds"] - def editor_url(self, lang, sid): + def locale_url(self, lang): return (f"https://crowdin.com/editor/{self.slug}/all/" - f"{self.source_code}-{self.editor_code.get(lang, lang)}#{sid}") + f"{self.source_code}-{self.editor_code.get(lang, lang)}") + + def editor_url(self, lang, sid): + return f"{self.locale_url(lang)}#{sid}" # ---- State ------------------------------------------------------------------- @@ -178,7 +185,7 @@ def slot_line(slot, project, suffix): def section_embeds(slots, project, title, color, suffix): - """One embed per locale, split when a locale outgrows a description.""" + """One embed per locale: up to LISTED_PER_LOCALE slots, then how many more.""" embeds = [] by_locale = collections.defaultdict(list) for slot in slots: @@ -186,17 +193,13 @@ def section_embeds(slots, project, title, color, suffix): for lang in sorted(by_locale, key=lambda lang: (-len(by_locale[lang]), lang)): items = sorted(by_locale[lang], key=lambda s: (s["identifier"] or "", str(s["pluralCategory"]))) - lines, used, first = [], 0, True - for slot in items: - line = slot_line(slot, project, suffix) - if lines and used + discord.text_len(line) + 1 > discord.MAX_EMBED_DESCRIPTION_CHARS: - embeds.append({"title": title(lang, len(items)) if first else f"{lang} (cont.)", - "description": "\n".join(lines), "color": color}) - lines, used, first = [], 0, False - lines.append(line) - used += discord.text_len(line) + 1 - embeds.append({"title": title(lang, len(items)) if first else f"{lang} (cont.)", - "description": "\n".join(lines), "color": color}) + listed = items if len(items) <= LISTED_PER_LOCALE else items[:LISTED_PER_LOCALE - 1] + lines = [slot_line(slot, project, suffix) for slot in listed] + if len(listed) < len(items): + lines.append(f"…and **{len(items) - len(listed)}** more: " + f"[open {lang} in the editor]({project.locale_url(lang)})") + embeds.append({"title": title(lang, len(items)), "description": "\n".join(lines), + "color": color}) return embeds diff --git a/tests/crowdin/test_duplicates.py b/tests/crowdin/test_duplicates.py index ef0bdfa..bd62d3a 100644 --- a/tests/crowdin/test_duplicates.py +++ b/tests/crowdin/test_duplicates.py @@ -72,16 +72,35 @@ def test_new_and_resolved_are_listed_with_editor_links(self): embeds[1]["description"]) self.assertEqual(embeds[2]["title"], "de — 1 resolved") - def test_a_long_locale_splits_across_embeds_and_messages(self): - opened = [finding(i, identifier="x" * 80) for i in range(400)] - messages = duplicates.build_messages(opened, [], 400, PROJECT) - embeds = [e for m in messages for e in m["embeds"]] - self.assertTrue(all(len(e["description"]) <= discord.MAX_EMBED_DESCRIPTION_CHARS - for e in embeds if "description" in e)) + def test_a_burst_in_one_locale_is_one_embed_with_a_count(self): + opened = [finding(i, identifier="x" * 80) for i in range(462)] + messages = duplicates.build_messages(opened, [], 462, PROJECT) + self.assertEqual(len(messages), 1) + locale = messages[0]["embeds"][1] + self.assertEqual(locale["title"], "de — 462 new") + lines = locale["description"].split("\n") + self.assertEqual(len(lines), duplicates.LISTED_PER_LOCALE) + self.assertEqual(lines[-1], f"…and **{462 - duplicates.LISTED_PER_LOCALE + 1}** more: " + "[open de in the editor](https://crowdin.com/editor/p/all/en-de)") + + def test_a_locale_at_the_limit_is_listed_in_full(self): + opened = [finding(i) for i in range(duplicates.LISTED_PER_LOCALE)] + description = duplicates.build_messages(opened, [], 9, PROJECT)[0]["embeds"][1]["description"] + self.assertEqual(description.count("• "), duplicates.LISTED_PER_LOCALE) + self.assertNotIn("more", description) + + def test_a_burst_across_every_locale_stays_within_discords_limits(self): + langs = [f"l{n}" for n in range(80)] + project = duplicates.Project({ + "identifier": "p", "sourceLanguage": {"id": "en"}, "targetLanguageIds": langs, + "targetLanguages": [{"id": lang} for lang in langs]}) + opened = [finding(i, lang=lang, identifier="x" * 90) for lang in langs + for i in range(50)] + messages = duplicates.build_messages(opened, [], len(opened), project) self.assertTrue(all(sum(discord.embed_len(e) for e in m["embeds"]) <= discord.MAX_EMBEDS_TEXT_CHARS for m in messages)) self.assertTrue(all(len(m["embeds"]) <= discord.MAX_EMBEDS_PER_MESSAGE for m in messages)) - self.assertEqual(sum(e["description"].count("\n• ") + 1 for e in embeds[1:]), 400) + self.assertEqual(len([e for m in messages for e in m["embeds"]]), 81) def recording(changes=None):