Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/jobs/crowdin-duplicates.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,8 @@ for a plural string, is what gets exported. A slot is (string, locale, plural ca
The open slots live in `/var/lib/session-ops/crowdin-duplicates/duplicates.json`. Each run posts
only what changed: slots newly holding 2+ translations, and slots that no longer do.
Nothing changed, nothing is posted.
Each locale is one embed listing up to 10 slots with editor links; past that, it gives
the count and a link to the locale in the editor, so a bulk import stays one message.

Reconciliation judges every string of every locale each weekday, after that day's
export, so a new duplicate is posted before the next one. A slot whose string was
Expand Down
31 changes: 17 additions & 14 deletions src/session_ops/crowdin/duplicates.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,10 @@

STATE_VERSION = 1
OPEN_COLOR, RESOLVED_COLOR = 0xE67E22, 0x2ECC71
# A locale's slots listed one by one; past this, a count and the locale's editor. A burst
# (a bulk import, a whole locale reviewed at once) then stays a message, not a flood that
# outruns Discord's per-channel rate limit.
LISTED_PER_LOCALE = 10


def user_label(u):
Expand Down Expand Up @@ -79,9 +83,12 @@ def __init__(self, details):
for lang in details["targetLanguages"]}
self.locales = details["targetLanguageIds"]

def editor_url(self, lang, sid):
def locale_url(self, lang):
return (f"https://crowdin.com/editor/{self.slug}/all/"
f"{self.source_code}-{self.editor_code.get(lang, lang)}#{sid}")
f"{self.source_code}-{self.editor_code.get(lang, lang)}")

def editor_url(self, lang, sid):
return f"{self.locale_url(lang)}#{sid}"


# ---- State -------------------------------------------------------------------
Expand Down Expand Up @@ -178,25 +185,21 @@ def slot_line(slot, project, suffix):


def section_embeds(slots, project, title, color, suffix):
"""One embed per locale, split when a locale outgrows a description."""
"""One embed per locale: up to LISTED_PER_LOCALE slots, then how many more."""
embeds = []
by_locale = collections.defaultdict(list)
for slot in slots:
by_locale[slot["locale"]].append(slot)
for lang in sorted(by_locale, key=lambda lang: (-len(by_locale[lang]), lang)):
items = sorted(by_locale[lang], key=lambda s: (s["identifier"] or "",
str(s["pluralCategory"])))
lines, used, first = [], 0, True
for slot in items:
line = slot_line(slot, project, suffix)
if lines and used + discord.text_len(line) + 1 > discord.MAX_EMBED_DESCRIPTION_CHARS:
embeds.append({"title": title(lang, len(items)) if first else f"{lang} (cont.)",
"description": "\n".join(lines), "color": color})
lines, used, first = [], 0, False
lines.append(line)
used += discord.text_len(line) + 1
embeds.append({"title": title(lang, len(items)) if first else f"{lang} (cont.)",
"description": "\n".join(lines), "color": color})
listed = items if len(items) <= LISTED_PER_LOCALE else items[:LISTED_PER_LOCALE - 1]
lines = [slot_line(slot, project, suffix) for slot in listed]
if len(listed) < len(items):
lines.append(f"…and **{len(items) - len(listed)}** more: "
f"[open {lang} in the editor]({project.locale_url(lang)})")
embeds.append({"title": title(lang, len(items)), "description": "\n".join(lines),
"color": color})
return embeds


Expand Down
33 changes: 26 additions & 7 deletions tests/crowdin/test_duplicates.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,16 +72,35 @@ def test_new_and_resolved_are_listed_with_editor_links(self):
embeds[1]["description"])
self.assertEqual(embeds[2]["title"], "de — 1 resolved")

def test_a_long_locale_splits_across_embeds_and_messages(self):
opened = [finding(i, identifier="x" * 80) for i in range(400)]
messages = duplicates.build_messages(opened, [], 400, PROJECT)
embeds = [e for m in messages for e in m["embeds"]]
self.assertTrue(all(len(e["description"]) <= discord.MAX_EMBED_DESCRIPTION_CHARS
for e in embeds if "description" in e))
def test_a_burst_in_one_locale_is_one_embed_with_a_count(self):
opened = [finding(i, identifier="x" * 80) for i in range(462)]
messages = duplicates.build_messages(opened, [], 462, PROJECT)
self.assertEqual(len(messages), 1)
locale = messages[0]["embeds"][1]
self.assertEqual(locale["title"], "de — 462 new")
lines = locale["description"].split("\n")
self.assertEqual(len(lines), duplicates.LISTED_PER_LOCALE)
self.assertEqual(lines[-1], f"…and **{462 - duplicates.LISTED_PER_LOCALE + 1}** more: "
"[open de in the editor](https://crowdin.com/editor/p/all/en-de)")

def test_a_locale_at_the_limit_is_listed_in_full(self):
opened = [finding(i) for i in range(duplicates.LISTED_PER_LOCALE)]
description = duplicates.build_messages(opened, [], 9, PROJECT)[0]["embeds"][1]["description"]
self.assertEqual(description.count("• "), duplicates.LISTED_PER_LOCALE)
self.assertNotIn("more", description)

def test_a_burst_across_every_locale_stays_within_discords_limits(self):
langs = [f"l{n}" for n in range(80)]
project = duplicates.Project({
"identifier": "p", "sourceLanguage": {"id": "en"}, "targetLanguageIds": langs,
"targetLanguages": [{"id": lang} for lang in langs]})
opened = [finding(i, lang=lang, identifier="x" * 90) for lang in langs
for i in range(50)]
messages = duplicates.build_messages(opened, [], len(opened), project)
self.assertTrue(all(sum(discord.embed_len(e) for e in m["embeds"])
<= discord.MAX_EMBEDS_TEXT_CHARS for m in messages))
self.assertTrue(all(len(m["embeds"]) <= discord.MAX_EMBEDS_PER_MESSAGE for m in messages))
self.assertEqual(sum(e["description"].count("\n• ") + 1 for e in embeds[1:]), 400)
self.assertEqual(len([e for m in messages for e in m["embeds"]]), 81)


def recording(changes=None):
Expand Down
Loading