mirror of
https://github.com/stenzek/duckstation.git
synced 2026-10-11 22:49:58 +00:00
Scripts: Translation tools experiments
This commit is contained in:
@@ -4,12 +4,14 @@ on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'src/duckstation-qt/translations/*.ts'
|
||||
- 'scripts/translation/**'
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- dev
|
||||
paths:
|
||||
- 'src/duckstation-qt/translations/*.ts'
|
||||
- 'scripts/translation/**'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
@@ -24,6 +26,9 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Test Translation Tools
|
||||
run: python -m unittest discover -s scripts/translation/tests -v
|
||||
|
||||
- name: Check Translation Placeholders
|
||||
shell: bash
|
||||
env:
|
||||
|
||||
+1
-1
@@ -45,7 +45,7 @@ CMakeLists.txt.user
|
||||
|
||||
# python bytecode
|
||||
__pycache__
|
||||
/.translation-work/
|
||||
|
||||
# other repos
|
||||
/android
|
||||
|
||||
|
||||
@@ -1,163 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Safely apply completed JSONL translation batches to a Qt TS catalog."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
if __package__ in (None, ""):
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from translation.ts_utils import ( # noqa: E402
|
||||
SCHEMA,
|
||||
MessageIdentity,
|
||||
catalog_fingerprint,
|
||||
load_jsonl,
|
||||
parse_catalog,
|
||||
placeholders_match,
|
||||
replace_translation_node,
|
||||
scan_raw_messages,
|
||||
validate_translation,
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("catalog", type=Path, help="Qt Linguist .ts catalog")
|
||||
parser.add_argument("batches", nargs="+", type=Path, help="completed JSONL batch files")
|
||||
parser.add_argument("--write", action="store_true", help="atomically update the catalog; default is dry-run")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def record_identity(record: dict[str, object]) -> MessageIdentity:
|
||||
return MessageIdentity(
|
||||
context=str(record.get("context", "")),
|
||||
source=str(record.get("source", "")),
|
||||
comment=str(record.get("comment", "")),
|
||||
extracomment=str(record.get("extracomment", "")),
|
||||
numerus=bool(record.get("numerus", False)),
|
||||
occurrence=int(record.get("occurrence", 0)),
|
||||
)
|
||||
|
||||
|
||||
def resolve_target(record: dict[str, object]) -> tuple[str | None, list[str] | None] | None:
|
||||
singular = record.get("target_translation")
|
||||
plurals = record.get("target_plural_translations")
|
||||
accept_current = record.get("accept_current") is True
|
||||
if accept_current:
|
||||
if singular is not None or plurals is not None:
|
||||
raise ValueError(f"{record.get('id')}: accept_current cannot be combined with target fields")
|
||||
current_plurals = record.get("current_plural_translations")
|
||||
if current_plurals:
|
||||
plurals = current_plurals
|
||||
else:
|
||||
singular = record.get("current_translation")
|
||||
|
||||
if singular is None and plurals is None:
|
||||
return None
|
||||
if singular is not None and plurals is not None:
|
||||
raise ValueError(f"{record.get('id')}: set either singular or plural target, not both")
|
||||
if singular is not None:
|
||||
if not isinstance(singular, str) or not singular.strip():
|
||||
raise ValueError(f"{record.get('id')}: target_translation must be a nonempty string")
|
||||
return singular, None
|
||||
if (
|
||||
not isinstance(plurals, list)
|
||||
or not plurals
|
||||
or any(not isinstance(value, str) or not value.strip() for value in plurals)
|
||||
):
|
||||
raise ValueError(f"{record.get('id')}: target_plural_translations must contain nonempty strings")
|
||||
return None, plurals
|
||||
|
||||
|
||||
def atomic_write(path: Path, text: str) -> None:
|
||||
mode = path.stat().st_mode
|
||||
temporary_name: str | None = None
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile("w", encoding="utf-8", newline="", dir=path.parent, delete=False) as stream:
|
||||
temporary_name = stream.name
|
||||
stream.write(text)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
os.chmod(temporary_name, mode)
|
||||
os.replace(temporary_name, path)
|
||||
finally:
|
||||
if temporary_name and os.path.exists(temporary_name):
|
||||
os.unlink(temporary_name)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
metadata, records = load_jsonl(args.batches)
|
||||
if len(metadata) != len(args.batches):
|
||||
raise SystemExit("each batch must contain exactly one metadata record")
|
||||
for item in metadata:
|
||||
if item.get("schema") != SCHEMA:
|
||||
raise SystemExit(f"unsupported batch schema: {item.get('schema')!r}")
|
||||
|
||||
_, messages = parse_catalog(args.catalog)
|
||||
fingerprint = catalog_fingerprint(messages)
|
||||
expected_fingerprints = {str(item.get("catalog_fingerprint", "")) for item in metadata}
|
||||
if expected_fingerprints != {fingerprint}:
|
||||
raise SystemExit("catalog source identity has changed since extraction; export fresh batches")
|
||||
|
||||
with args.catalog.open("r", encoding="utf-8", newline="") as stream:
|
||||
raw_text = stream.read()
|
||||
spans = scan_raw_messages(raw_text)
|
||||
seen: set[str] = set()
|
||||
replacements: list[tuple[int, int, str]] = []
|
||||
skipped = 0
|
||||
for record in records:
|
||||
identifier = str(record.get("id", ""))
|
||||
identity = record_identity(record)
|
||||
if identifier != identity.identifier:
|
||||
raise SystemExit(f"{identifier or '<missing id>'}: record identity does not match its id")
|
||||
if identifier in seen:
|
||||
raise SystemExit(f"duplicate record id across batches: {identifier}")
|
||||
seen.add(identifier)
|
||||
target = resolve_target(record)
|
||||
if target is None:
|
||||
skipped += 1
|
||||
continue
|
||||
span = spans.get(identifier)
|
||||
if span is None:
|
||||
raise SystemExit(f"message no longer exists in catalog: {identifier}")
|
||||
if span.translation_type != "unfinished":
|
||||
raise SystemExit(f"refusing to overwrite {span.translation_type} translation: {identifier}")
|
||||
singular, plurals = target
|
||||
if span.numerus != (plurals is not None):
|
||||
raise SystemExit(f"singular/plural target does not match catalog message: {identifier}")
|
||||
values = plurals if plurals is not None else [singular or ""]
|
||||
for value in values:
|
||||
problems = validate_translation(
|
||||
identity.source,
|
||||
value,
|
||||
allow_missing_placeholders=plurals is not None,
|
||||
)
|
||||
if problems:
|
||||
raise SystemExit(f"{identifier}: {'; '.join(problems)}")
|
||||
if plurals is not None and not any(placeholders_match(identity.source, value) for value in plurals):
|
||||
raise SystemExit(f"{identifier}: no plural form preserves the complete source placeholder set")
|
||||
replacement_block = replace_translation_node(span.block, singular, plurals)
|
||||
replacements.append((span.start, span.end, replacement_block))
|
||||
|
||||
updated_text = raw_text
|
||||
for start, end, replacement in sorted(replacements, reverse=True):
|
||||
updated_text = updated_text[:start] + replacement + updated_text[end:]
|
||||
if args.write and replacements:
|
||||
atomic_write(args.catalog, updated_text)
|
||||
|
||||
action = "Applied" if args.write else "Would apply"
|
||||
print(f"{action} {len(replacements)} translation(s); skipped {skipped} record(s) without targets.")
|
||||
if not args.write:
|
||||
print("Dry-run only; pass --write to update the catalog.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,179 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Export active unfinished Qt TS messages to editable JSONL batches."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import collections
|
||||
import difflib
|
||||
from pathlib import Path
|
||||
|
||||
if __package__ in (None, ""):
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from translation.ts_utils import ( # noqa: E402
|
||||
SCHEMA,
|
||||
SKIPPED_TYPES,
|
||||
CatalogMessage,
|
||||
catalog_fingerprint,
|
||||
message_to_batch_record,
|
||||
parse_catalog,
|
||||
write_jsonl,
|
||||
)
|
||||
|
||||
|
||||
def translated_text(message: CatalogMessage) -> str | list[str] | None:
|
||||
if message.plural_translations:
|
||||
return message.plural_translations if any(value.strip() for value in message.plural_translations) else None
|
||||
return message.translation if message.translation and message.translation.strip() else None
|
||||
|
||||
|
||||
def build_suggestions(
|
||||
target: CatalogMessage,
|
||||
messages: list[CatalogMessage],
|
||||
limit: int,
|
||||
minimum_similarity: float,
|
||||
) -> list[dict[str, object]]:
|
||||
if limit <= 0:
|
||||
return []
|
||||
candidates: list[tuple[float, CatalogMessage]] = []
|
||||
for candidate in messages:
|
||||
text = translated_text(candidate)
|
||||
if text is None or candidate.identifier == target.identifier:
|
||||
continue
|
||||
exact = candidate.identity.source == target.identity.source
|
||||
if not exact and candidate.identity.context != target.identity.context:
|
||||
continue
|
||||
similarity = (
|
||||
1.0
|
||||
if exact
|
||||
else difflib.SequenceMatcher(None, target.identity.source, candidate.identity.source).ratio()
|
||||
)
|
||||
if similarity >= minimum_similarity:
|
||||
candidates.append((similarity, candidate))
|
||||
candidates.sort(
|
||||
key=lambda pair: (
|
||||
pair[0],
|
||||
pair[1].translation_type == "finished",
|
||||
pair[1].identity.source == target.identity.source,
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
output: list[dict[str, object]] = []
|
||||
seen: set[tuple[str, str]] = set()
|
||||
for similarity, candidate in candidates:
|
||||
text = translated_text(candidate)
|
||||
signature = (candidate.identity.source, repr(text))
|
||||
if signature in seen:
|
||||
continue
|
||||
seen.add(signature)
|
||||
output.append(
|
||||
{
|
||||
"source": candidate.identity.source,
|
||||
"translation": text,
|
||||
"context": candidate.identity.context,
|
||||
"type": candidate.translation_type,
|
||||
"similarity": round(similarity, 4),
|
||||
}
|
||||
)
|
||||
if len(output) == limit:
|
||||
break
|
||||
return output
|
||||
|
||||
|
||||
def split_balanced(records: list[dict[str, object]], batch_count: int) -> list[list[dict[str, object]]]:
|
||||
if batch_count <= 1 or len(records) <= 1:
|
||||
return [records]
|
||||
batch_count = min(batch_count, len(records))
|
||||
by_context: dict[str, list[dict[str, object]]] = collections.defaultdict(list)
|
||||
for record in records:
|
||||
by_context[str(record["context"])].append(record)
|
||||
|
||||
target_size = max(1, (len(records) + batch_count - 1) // batch_count)
|
||||
chunks: list[list[dict[str, object]]] = []
|
||||
for context_records in by_context.values():
|
||||
if len(context_records) > target_size * 3 // 2:
|
||||
chunks.extend(
|
||||
context_records[offset : offset + target_size]
|
||||
for offset in range(0, len(context_records), target_size)
|
||||
)
|
||||
else:
|
||||
chunks.append(context_records)
|
||||
|
||||
batches: list[list[dict[str, object]]] = [[] for _ in range(batch_count)]
|
||||
for chunk in sorted(chunks, key=len, reverse=True):
|
||||
destination = min(batches, key=len)
|
||||
destination.extend(chunk)
|
||||
return batches
|
||||
|
||||
|
||||
def make_metadata(catalog: Path, fingerprint: str, batch_index: int, batch_count: int) -> dict[str, object]:
|
||||
return {
|
||||
"record_type": "metadata",
|
||||
"schema": SCHEMA,
|
||||
"catalog": str(catalog),
|
||||
"catalog_fingerprint": fingerprint,
|
||||
"batch_index": batch_index,
|
||||
"batch_count": batch_count,
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("catalog", type=Path, help="Qt Linguist .ts catalog")
|
||||
output = parser.add_mutually_exclusive_group(required=True)
|
||||
output.add_argument("--output", type=Path, help="single JSONL output file")
|
||||
output.add_argument("--batch-dir", type=Path, help="directory for numbered JSONL batches")
|
||||
parser.add_argument("--batches", type=int, default=1, help="number of balanced batches (with --batch-dir)")
|
||||
parser.add_argument("--context", action="append", default=[], help="include only this context; repeatable")
|
||||
parser.add_argument("--suggestions", type=int, default=2, help="translation-memory suggestions per message")
|
||||
parser.add_argument("--similarity", type=float, default=0.72, help="minimum fuzzy suggestion similarity")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
if args.output and args.batches != 1:
|
||||
raise SystemExit("--batches requires --batch-dir")
|
||||
if args.batches < 1:
|
||||
raise SystemExit("--batches must be at least 1")
|
||||
if not 0.0 <= args.similarity <= 1.0:
|
||||
raise SystemExit("--similarity must be between 0 and 1")
|
||||
|
||||
_, messages = parse_catalog(args.catalog)
|
||||
contexts = set(args.context)
|
||||
targets = [
|
||||
message
|
||||
for message in messages
|
||||
if message.translation_type == "unfinished"
|
||||
and message.translation_type not in SKIPPED_TYPES
|
||||
and (not contexts or message.identity.context in contexts)
|
||||
]
|
||||
records = []
|
||||
for message in targets:
|
||||
record = message_to_batch_record(message)
|
||||
record["suggestions"] = build_suggestions(message, messages, args.suggestions, args.similarity)
|
||||
records.append(record)
|
||||
|
||||
fingerprint = catalog_fingerprint(messages)
|
||||
if args.output:
|
||||
write_jsonl(args.output, make_metadata(args.catalog, fingerprint, 1, 1), records)
|
||||
destinations = [args.output]
|
||||
else:
|
||||
batches = split_balanced(records, args.batches)
|
||||
destinations = []
|
||||
for index, batch in enumerate(batches, 1):
|
||||
destination = args.batch_dir / f"batch-{index:03d}.jsonl"
|
||||
write_jsonl(destination, make_metadata(args.catalog, fingerprint, index, len(batches)), batch)
|
||||
destinations.append(destination)
|
||||
|
||||
print(f"Exported {len(records)} active unfinished messages to {len(destinations)} file(s).")
|
||||
for destination in destinations:
|
||||
print(destination)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -11,8 +11,10 @@ from pathlib import Path
|
||||
|
||||
SCRIPTS_DIR = Path(__file__).resolve().parents[2]
|
||||
REPO_ROOT = SCRIPTS_DIR.parent
|
||||
TRANSLATE = SCRIPTS_DIR / "translation" / "translate_ts.py"
|
||||
sys.path.insert(0, str(SCRIPTS_DIR))
|
||||
|
||||
from translation.translate_ts import validate_decision # noqa: E402
|
||||
from translation.ts_utils import ( # noqa: E402
|
||||
TRANSLATION_RE,
|
||||
catalog_fingerprint,
|
||||
@@ -53,6 +55,10 @@ FIXTURE = """<?xml version="1.0" encoding="utf-8"?>
|
||||
<source>New wording</source>
|
||||
<translation type="unfinished"></translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>&Open</source>
|
||||
<translation type="unfinished"></translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Old wording</source>
|
||||
<translation type="vanished">古い文言</translation>
|
||||
@@ -78,9 +84,9 @@ FIXTURE = """<?xml version="1.0" encoding="utf-8"?>
|
||||
|
||||
|
||||
class TranslationToolTests(unittest.TestCase):
|
||||
def run_tool(self, script: str, *arguments: object, expect: int = 0) -> subprocess.CompletedProcess[str]:
|
||||
def run_tool(self, *arguments: object, expect: int = 0) -> subprocess.CompletedProcess[str]:
|
||||
result = subprocess.run(
|
||||
[sys.executable, str(SCRIPTS_DIR / "translation" / script), *(str(value) for value in arguments)],
|
||||
[sys.executable, str(TRANSLATE), *(str(value) for value in arguments)],
|
||||
cwd=REPO_ROOT,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
@@ -90,7 +96,76 @@ class TranslationToolTests(unittest.TestCase):
|
||||
self.assertEqual(expect, result.returncode, result.stdout)
|
||||
return result
|
||||
|
||||
def test_placeholder_recognition_avoids_percent_and_escaped_brace_false_positives(self) -> None:
|
||||
def run_validator(self, *arguments: object, expect: int = 0) -> subprocess.CompletedProcess[str]:
|
||||
result = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(SCRIPTS_DIR / "translation" / "validate_ts.py"),
|
||||
*(str(value) for value in arguments),
|
||||
],
|
||||
cwd=REPO_ROOT,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
check=False,
|
||||
)
|
||||
self.assertEqual(expect, result.returncode, result.stdout)
|
||||
return result
|
||||
|
||||
def start_session(
|
||||
self, root: Path, fixture: str = FIXTURE, *extra: object
|
||||
) -> tuple[Path, Path, dict[str, object]]:
|
||||
catalog = root / "catalog.ts"
|
||||
session = root / "session"
|
||||
catalog.write_text(fixture, encoding="utf-8")
|
||||
self.run_tool("start", catalog, "--session-dir", session, *extra)
|
||||
manifest = json.loads((session / "manifest.json").read_text(encoding="utf-8"))
|
||||
return catalog, session, manifest
|
||||
|
||||
def task_records(self, session: Path, task: dict[str, object]) -> list[dict[str, object]]:
|
||||
lines = [
|
||||
json.loads(line)
|
||||
for line in (session / str(task["path"])).read_text(encoding="utf-8").splitlines()
|
||||
]
|
||||
return lines[1:]
|
||||
|
||||
def response_path(self, session: Path, task: dict[str, object]) -> Path:
|
||||
return session / str(task["response_path"])
|
||||
|
||||
def append_decisions(
|
||||
self, response: Path, decisions: list[dict[str, object]], malformed_tail: str = ""
|
||||
) -> None:
|
||||
text = response.read_text(encoding="utf-8")
|
||||
text += "".join(json.dumps(item, ensure_ascii=False, sort_keys=True) + "\n" for item in decisions)
|
||||
response.write_text(text + malformed_tail, encoding="utf-8")
|
||||
|
||||
def decision_for(self, record: dict[str, object]) -> dict[str, object]:
|
||||
identifier = record["id"]
|
||||
source = record["source"]
|
||||
context = record["context"]
|
||||
if source == "Hello %1 {0} ${title} %.1f <strong>world</strong>":
|
||||
return {
|
||||
"id": identifier,
|
||||
"translation": "こんにちは %1 {0} ${title} %.1f <strong>世界</strong>",
|
||||
}
|
||||
if source == "%n file(s)":
|
||||
return {"id": identifier, "plural_translations": ["%n ファイル"]}
|
||||
if source == "New wording":
|
||||
return {"id": identifier, "translation": "新しい文言"}
|
||||
if source == "&Open":
|
||||
return {"id": identifier, "translation": "開く(&O)"}
|
||||
if source == "Register" and context == "Alpha":
|
||||
return {"id": identifier, "translation": "登録"}
|
||||
if source == "Register":
|
||||
return {"id": identifier, "translation": "登録"}
|
||||
raise AssertionError((context, source))
|
||||
|
||||
def complete_responses(self, session: Path, manifest: dict[str, object]) -> None:
|
||||
for task in manifest["tasks"]:
|
||||
decisions = [self.decision_for(record) for record in self.task_records(session, task)]
|
||||
self.append_decisions(self.response_path(session, task), decisions)
|
||||
|
||||
def test_placeholder_recognition_and_compatibility(self) -> None:
|
||||
text = "At 100% speed, 5% of users: %1 %n {} {0:08X} %.1f %s ${title} {{}} %%"
|
||||
placeholders = extract_placeholders(text)
|
||||
self.assertEqual(1, placeholders["qt:%1"])
|
||||
@@ -105,37 +180,28 @@ class TranslationToolTests(unittest.TestCase):
|
||||
self.assertFalse(placeholders_match("{} {}", "{0} {0}"))
|
||||
self.assertTrue(placeholders_match("Use {0}, then use {0} again", "Usar {0}"))
|
||||
self.assertFalse(placeholder_counts_match("Use {0}, then use {0} again", "Usar {0}"))
|
||||
self.assertTrue(placeholders_match("Saved at {0:%H:%M} on {0:%Y/%m/%d}", "Guardado {0:%d/%m/%Y}"))
|
||||
self.assertFalse(
|
||||
placeholder_counts_match("Saved at {0:%H:%M} on {0:%Y/%m/%d}", "Guardado {0:%d/%m/%Y}")
|
||||
)
|
||||
self.assertFalse(placeholder_counts_match("{0} {0} {1}", "{0} {1} {1}"))
|
||||
self.assertFalse(placeholders_match("Use %1 and %2", "Usar %1"))
|
||||
self.assertTrue(placeholders_are_subset("{} of %n", "%n"))
|
||||
self.assertFalse(placeholders_are_subset("{} of %n", "%n %1"))
|
||||
self.assertEqual({"qt:%1": 1}, dict(extract_placeholders("%1x")))
|
||||
|
||||
def test_rich_text_ignores_angle_bracket_labels(self) -> None:
|
||||
def test_rich_text_helpers_ignore_labels_and_find_unbalanced_tags(self) -> None:
|
||||
self.assertFalse(extract_rich_tags("<Parent Directory>"))
|
||||
self.assertEqual({"strong": 1, "/strong": 1}, dict(extract_rich_tags("<strong>Text</strong>")))
|
||||
self.assertFalse(validate_translation("<strong>Text</strong>", "テキスト"))
|
||||
|
||||
def test_rich_text_tag_pairs_are_balanced(self) -> None:
|
||||
self.assertFalse(unbalanced_rich_tags("<html><head/><body><br><hr/></body></html>"))
|
||||
self.assertFalse(unbalanced_rich_tags("<strong><b>Text</strong></b>"))
|
||||
self.assertEqual({"p": (2, 1), "strong": (1, 0)}, unbalanced_rich_tags("<p><p><strong>Text</p>"))
|
||||
|
||||
def test_fingerprint_ignores_translation_changes(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
first = Path(directory) / "first.ts"
|
||||
second = Path(directory) / "second.ts"
|
||||
root = Path(directory)
|
||||
first = root / "first.ts"
|
||||
second = root / "second.ts"
|
||||
first.write_text(FIXTURE, encoding="utf-8")
|
||||
second.write_text(FIXTURE.replace("レジスタ", "登録"), encoding="utf-8")
|
||||
_, first_messages = parse_catalog(first)
|
||||
_, second_messages = parse_catalog(second)
|
||||
self.assertEqual(catalog_fingerprint(first_messages), catalog_fingerprint(second_messages))
|
||||
|
||||
def test_translation_node_replacement_preserves_message_text(self) -> None:
|
||||
def test_translation_replacement_preserves_source_and_can_insert_missing_node(self) -> None:
|
||||
block = (
|
||||
" <message>\n"
|
||||
" <source>A & B</source>\n"
|
||||
@@ -146,47 +212,211 @@ class TranslationToolTests(unittest.TestCase):
|
||||
self.assertIn("<source>A & B</source>", replaced)
|
||||
self.assertIn("<translation>A と B</translation>", replaced)
|
||||
self.assertNotIn("unfinished", replaced)
|
||||
missing = " <message>\n <source>Hello</source>\n </message>"
|
||||
inserted = replace_translation_node(missing, "こんにちは", None)
|
||||
self.assertIn(" <translation>こんにちは</translation>\n </message>", inserted)
|
||||
|
||||
def test_end_to_end_extract_apply_validate(self) -> None:
|
||||
def test_start_creates_bounded_immutable_tasks_without_old_messages(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog = root / "catalog.ts"
|
||||
batch = root / "batch.jsonl"
|
||||
catalog.write_text(FIXTURE, encoding="utf-8")
|
||||
original = catalog.read_bytes()
|
||||
_, session, manifest = self.start_session(Path(directory), FIXTURE, "--max-records", 2)
|
||||
self.assertEqual(6, manifest["target_count"])
|
||||
self.assertEqual(3, len(manifest["tasks"]))
|
||||
sources = []
|
||||
for task in manifest["tasks"]:
|
||||
records = self.task_records(session, task)
|
||||
self.assertLessEqual(len(records), 2)
|
||||
sources.extend(record["source"] for record in records)
|
||||
self.assertNotIn("Old wording", sources)
|
||||
self.assertNotIn("Removed", sources)
|
||||
|
||||
self.run_tool("extract_ts.py", catalog, "--output", batch, "--suggestions", 3)
|
||||
lines = [json.loads(line) for line in batch.read_text(encoding="utf-8").splitlines()]
|
||||
records = [line for line in lines if line["record_type"] == "message"]
|
||||
self.assertEqual(5, len(records))
|
||||
self.assertNotIn("Old wording", {record["source"] for record in records})
|
||||
new_wording = next(record for record in records if record["source"] == "New wording")
|
||||
self.assertTrue(any(item["type"] == "vanished" for item in new_wording["suggestions"]))
|
||||
status = self.run_tool("status", session)
|
||||
self.assertIn("INCOMPLETE: dispatch the next bounded tasks", status.stdout)
|
||||
self.assertIn("Concurrent agent slots are not a total-capacity limit", status.stdout)
|
||||
|
||||
translations: dict[tuple[str, str], str | list[str]] = {
|
||||
(
|
||||
"Alpha",
|
||||
"Hello %1 {0} ${title} %.1f <strong>world</strong>",
|
||||
): "こんにちは %1 {0} ${title} %.1f <strong>世界</strong>",
|
||||
("Alpha", "Register"): "登録",
|
||||
("Alpha", "%n file(s)"): ["%n ファイル"],
|
||||
("Alpha", "New wording"): "新しい文言",
|
||||
("Beta", "Register"): "登録",
|
||||
}
|
||||
for record in records:
|
||||
target = translations[(record["context"], record["source"])]
|
||||
if isinstance(target, list):
|
||||
record["target_plural_translations"] = target
|
||||
else:
|
||||
record["target_translation"] = target
|
||||
batch.write_text(
|
||||
"\n".join(json.dumps(line, ensure_ascii=False) for line in lines) + "\n",
|
||||
task_path = session / manifest["tasks"][0]["path"]
|
||||
task_path.write_text(task_path.read_text(encoding="utf-8") + "\n", encoding="utf-8")
|
||||
result = self.run_tool("status", session, expect=1)
|
||||
self.assertIn("checksum mismatch", result.stdout)
|
||||
|
||||
def test_large_catalog_scales_by_adding_bounded_tasks(self) -> None:
|
||||
messages = "".join(
|
||||
" <message>\n"
|
||||
f" <source>Message {index} with some source text</source>\n"
|
||||
" <translation type=\"unfinished\"></translation>\n"
|
||||
" </message>\n"
|
||||
for index in range(2500)
|
||||
)
|
||||
fixture = (
|
||||
'<?xml version="1.0" encoding="utf-8"?>\n<!DOCTYPE TS>\n'
|
||||
'<TS version="2.1" language="ja">\n<context>\n <name>Large</name>\n'
|
||||
f"{messages}</context>\n</TS>\n"
|
||||
)
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(
|
||||
Path(directory),
|
||||
fixture,
|
||||
"--suggestions",
|
||||
0,
|
||||
)
|
||||
self.assertEqual(9, len(manifest["tasks"]))
|
||||
self.assertEqual(2500, manifest["target_count"])
|
||||
for index, task in enumerate(manifest["tasks"]):
|
||||
records = self.task_records(session, task)
|
||||
expected_count = 100 if index == 8 else 300
|
||||
self.assertEqual(expected_count, len(records))
|
||||
size = (session / task["path"]).stat().st_size
|
||||
self.assertLessEqual(size, 192 * 1024)
|
||||
|
||||
def test_partial_checkpoint_can_merge_and_resume(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(Path(directory))
|
||||
task = manifest["tasks"][0]
|
||||
records = self.task_records(session, task)
|
||||
response = self.response_path(session, task)
|
||||
self.append_decisions(response, [self.decision_for(record) for record in records[:2]])
|
||||
self.run_tool("check", session, response)
|
||||
self.run_tool("merge", session, "--write")
|
||||
status = self.run_tool("status", session, "--json")
|
||||
payload = json.loads(status.stdout)
|
||||
self.assertEqual(2, payload["reviewed"])
|
||||
self.assertEqual(4, payload["remaining"])
|
||||
self.assertEqual(2, payload["tasks"][0]["checkpointed"])
|
||||
self.assertEqual(0, payload["tasks"][0]["unmerged"])
|
||||
|
||||
self.append_decisions(response, [self.decision_for(record) for record in records[2:]])
|
||||
self.run_tool("check", session, response, "--require-complete-task")
|
||||
self.run_tool("merge", session, "--write")
|
||||
status = json.loads(self.run_tool("status", session, "--json").stdout)
|
||||
self.assertEqual(0, status["remaining"])
|
||||
|
||||
def test_merge_is_idempotent_and_conflicts_require_replace(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(Path(directory))
|
||||
task = manifest["tasks"][0]
|
||||
record = self.task_records(session, task)[0]
|
||||
response = self.response_path(session, task)
|
||||
decision = self.decision_for(record)
|
||||
self.append_decisions(response, [decision])
|
||||
self.run_tool("merge", session, "--write")
|
||||
repeated = self.run_tool("merge", session, "--write")
|
||||
self.assertIn("Merged 0", repeated.stdout)
|
||||
|
||||
lines = response.read_text(encoding="utf-8").splitlines()
|
||||
changed = {"id": record["id"], "translation": "別の %1 {0} ${title} %.1f <strong>訳</strong>"}
|
||||
response.write_text(lines[0] + "\n" + json.dumps(changed, ensure_ascii=False) + "\n", encoding="utf-8")
|
||||
status = json.loads(self.run_tool("status", session, "--json").stdout)
|
||||
self.assertEqual(1, status["unmerged"])
|
||||
result = self.run_tool("merge", session, response, "--write", expect=1)
|
||||
self.assertIn("conflicts with the merged review", result.stdout)
|
||||
self.run_tool("merge", session, response, "--replace", "--write")
|
||||
|
||||
def test_malformed_response_does_not_change_journal(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(Path(directory))
|
||||
response = self.response_path(session, manifest["tasks"][0])
|
||||
self.append_decisions(response, [], malformed_tail='{"id":')
|
||||
journal = session / "reviews.jsonl"
|
||||
original = journal.read_bytes()
|
||||
self.run_tool("merge", session, "--write", expect=1)
|
||||
self.assertEqual(original, journal.read_bytes())
|
||||
|
||||
def test_salvage_recovers_valid_decisions_from_damaged_response(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(Path(directory))
|
||||
task = manifest["tasks"][0]
|
||||
records = self.task_records(session, task)
|
||||
response = self.response_path(session, task)
|
||||
valid = self.decision_for(records[0])
|
||||
source_equal = next(record for record in records if record["source"] == "New wording")
|
||||
response.write_text(
|
||||
"{record_type: metadata, task_id: task-0001}\n"
|
||||
+ json.dumps(valid, ensure_ascii=False)
|
||||
+ "\n"
|
||||
+ json.dumps(
|
||||
{"id": source_equal["id"], "translation": source_equal["source"]},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
+ "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
original = response.read_bytes()
|
||||
|
||||
self.run_tool("apply_ts.py", catalog, batch)
|
||||
self.assertEqual(original, catalog.read_bytes(), "dry-run modified the catalog")
|
||||
self.run_tool("apply_ts.py", catalog, batch, "--write")
|
||||
status = self.run_tool("status", session, "--json", expect=1)
|
||||
payload = json.loads(status.stdout)
|
||||
self.assertEqual("salvage_responses", payload["recommended_action"])
|
||||
self.assertEqual([str(response.resolve())], payload["invalid_response_paths"])
|
||||
|
||||
dry_run = self.run_tool("salvage", session, response)
|
||||
self.assertIn("Would salvage 2/6 valid decision(s)", dry_run.stdout)
|
||||
self.assertIn("normalized source-equal translation", dry_run.stdout)
|
||||
self.assertEqual(original, response.read_bytes())
|
||||
|
||||
repaired = self.run_tool("salvage", session, response, "--write")
|
||||
self.assertIn("Salvaged 2/6 valid decision(s)", repaired.stdout)
|
||||
self.assertIn("skipped 1 invalid line(s)", repaired.stdout)
|
||||
self.run_tool("check", session, response)
|
||||
self.run_tool("merge", session, "--write")
|
||||
payload = json.loads(self.run_tool("status", session, "--json").stdout)
|
||||
self.assertEqual(2, payload["reviewed"])
|
||||
self.assertEqual(4, payload["remaining"])
|
||||
self.assertEqual("dispatch_workers", payload["recommended_action"])
|
||||
|
||||
def test_response_validation_normalizes_source_copy_and_rejects_plural_shape_tags_and_accelerators(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
_, session, manifest = self.start_session(Path(directory))
|
||||
records = [
|
||||
record
|
||||
for task in manifest["tasks"]
|
||||
for record in self.task_records(session, task)
|
||||
]
|
||||
by_source = {record["source"]: record for record in records}
|
||||
source_equal = by_source["New wording"]
|
||||
normalized, warnings = validate_decision(
|
||||
source_equal, {"id": source_equal["id"], "translation": "New wording"}
|
||||
)
|
||||
self.assertTrue(normalized["accept_source"])
|
||||
self.assertIn("normalized source-equal translation", warnings[0])
|
||||
source_equal_current = dict(source_equal)
|
||||
source_equal_current["current_translation"] = source_equal_current["source"]
|
||||
normalized, warnings = validate_decision(
|
||||
source_equal_current,
|
||||
{"id": source_equal["id"], "accept_current": True},
|
||||
)
|
||||
self.assertTrue(normalized["accept_source"])
|
||||
self.assertIn("normalized source-equal accept_current", warnings[0])
|
||||
normalized, _ = validate_decision(
|
||||
source_equal, {"id": source_equal["id"], "accept_source": True}
|
||||
)
|
||||
self.assertTrue(normalized["accept_source"])
|
||||
|
||||
plural = by_source["%n file(s)"]
|
||||
with self.assertRaisesRegex(ValueError, "exactly 1"):
|
||||
validate_decision(plural, {"id": plural["id"], "plural_translations": ["a", "b"]})
|
||||
rich = by_source["Hello %1 {0} ${title} %.1f <strong>world</strong>"]
|
||||
with self.assertRaisesRegex(ValueError, "rich-text"):
|
||||
validate_decision(
|
||||
rich,
|
||||
{
|
||||
"id": rich["id"],
|
||||
"translation": "こんにちは %1 {0} ${title} %.1f 世界",
|
||||
},
|
||||
)
|
||||
accelerator = by_source["&Open"]
|
||||
with self.assertRaisesRegex(ValueError, "accelerator"):
|
||||
validate_decision(accelerator, {"id": accelerator["id"], "translation": "開く"})
|
||||
|
||||
def test_complete_session_applies_atomically_and_is_idempotent(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog, session, manifest = self.start_session(root, FIXTURE, "--max-records", 2)
|
||||
original = catalog.read_bytes()
|
||||
self.complete_responses(session, manifest)
|
||||
self.run_tool("merge", session, "--write")
|
||||
status = self.run_tool("status", session)
|
||||
self.assertIn("REVIEW QUEUE COMPLETE", status.stdout)
|
||||
self.run_tool("apply", session)
|
||||
self.assertEqual(original, catalog.read_bytes())
|
||||
self.run_tool("apply", session, "--write")
|
||||
updated = catalog.read_text(encoding="utf-8")
|
||||
|
||||
def normalize_translations(text: str) -> str:
|
||||
@@ -196,67 +426,108 @@ class TranslationToolTests(unittest.TestCase):
|
||||
self.assertIn('<translation type="vanished">古い文言</translation>', updated)
|
||||
self.assertIn('<translation type="obsolete">削除済み</translation>', updated)
|
||||
self.assertNotIn('type="unfinished"', updated)
|
||||
self.run_tool("validate_ts.py", catalog, "--require-complete")
|
||||
repeated = self.run_tool("apply", session, "--write")
|
||||
self.assertIn("0 translation(s)", repeated.stdout)
|
||||
self.assertEqual(updated, catalog.read_text(encoding="utf-8"))
|
||||
self.run_validator(catalog, "--require-complete")
|
||||
|
||||
def test_accept_current_and_balanced_batches(self) -> None:
|
||||
def test_incomplete_or_stale_session_never_changes_catalog(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog = root / "catalog.ts"
|
||||
batches = root / "batches"
|
||||
catalog.write_text(FIXTURE, encoding="utf-8")
|
||||
self.run_tool("extract_ts.py", catalog, "--batch-dir", batches, "--batches", 3)
|
||||
files = sorted(batches.glob("*.jsonl"))
|
||||
self.assertEqual(3, len(files))
|
||||
found = False
|
||||
for path in files:
|
||||
lines = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()]
|
||||
for line in lines:
|
||||
if line.get("context") == "Alpha" and line.get("source") == "Register":
|
||||
line["accept_current"] = True
|
||||
found = True
|
||||
path.write_text(
|
||||
"\n".join(json.dumps(line, ensure_ascii=False) for line in lines) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
self.assertTrue(found)
|
||||
result = self.run_tool("apply_ts.py", catalog, *files)
|
||||
self.assertIn("Would apply 1 translation", result.stdout)
|
||||
catalog, session, manifest = self.start_session(root)
|
||||
original = catalog.read_bytes()
|
||||
self.run_tool("apply", session, "--write", expect=1)
|
||||
self.assertEqual(original, catalog.read_bytes())
|
||||
|
||||
def test_batches_apply_sequentially_without_invalidating_fingerprint(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog = root / "catalog.ts"
|
||||
batches = root / "batches"
|
||||
catalog.write_text(FIXTURE, encoding="utf-8")
|
||||
self.run_tool("extract_ts.py", catalog, "--batch-dir", batches, "--batches", 2)
|
||||
files = sorted(batches.glob("*.jsonl"))
|
||||
self.assertEqual(2, len(files))
|
||||
for path in files:
|
||||
lines = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()]
|
||||
record = next(line for line in lines if line["record_type"] == "message")
|
||||
if record["numerus"]:
|
||||
record["target_plural_translations"] = [record["source"]]
|
||||
else:
|
||||
record["target_translation"] = record["source"]
|
||||
path.write_text(
|
||||
"\n".join(json.dumps(line, ensure_ascii=False) for line in lines) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
self.run_tool("apply_ts.py", catalog, files[0], "--write")
|
||||
self.run_tool("apply_ts.py", catalog, files[1], "--write")
|
||||
|
||||
def test_stale_catalog_is_rejected(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog = root / "catalog.ts"
|
||||
batch = root / "batch.jsonl"
|
||||
catalog.write_text(FIXTURE, encoding="utf-8")
|
||||
self.run_tool("extract_ts.py", catalog, "--output", batch)
|
||||
self.complete_responses(session, manifest)
|
||||
self.run_tool("merge", session, "--write")
|
||||
catalog.write_text(FIXTURE.replace("New wording", "Changed source"), encoding="utf-8")
|
||||
result = self.run_tool("apply_ts.py", catalog, batch, expect=1)
|
||||
changed = catalog.read_bytes()
|
||||
result = self.run_tool("apply", session, "--write", expect=1)
|
||||
self.assertIn("source identity has changed", result.stdout)
|
||||
self.assertEqual(changed, catalog.read_bytes())
|
||||
|
||||
def test_validation_reports_catalog_and_source_lines(self) -> None:
|
||||
def test_reviewed_only_apply_atomically_harvests_partial_session(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog, session, manifest = self.start_session(root)
|
||||
original = catalog.read_bytes()
|
||||
task = manifest["tasks"][0]
|
||||
record = self.task_records(session, task)[0]
|
||||
self.append_decisions(self.response_path(session, task), [self.decision_for(record)])
|
||||
self.run_tool("merge", session, "--write")
|
||||
|
||||
dry_run = self.run_tool("apply", session, "--reviewed-only")
|
||||
self.assertIn("Would apply 1 translation(s)", dry_run.stdout)
|
||||
self.assertIn("5 remain unreviewed", dry_run.stdout)
|
||||
self.assertEqual(original, catalog.read_bytes())
|
||||
|
||||
applied = self.run_tool("apply", session, "--reviewed-only", "--write")
|
||||
self.assertIn("Applied 1 translation(s)", applied.stdout)
|
||||
updated = catalog.read_text(encoding="utf-8")
|
||||
self.assertIn(
|
||||
"<translation>こんにちは %1 {0} ${title} %.1f "
|
||||
"<strong>世界</strong></translation>",
|
||||
updated,
|
||||
)
|
||||
self.assertEqual(5, updated.count('type="unfinished"'))
|
||||
|
||||
def test_target_conflict_fails_but_unrelated_translation_edit_is_preserved(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog, session, manifest = self.start_session(root)
|
||||
self.complete_responses(session, manifest)
|
||||
self.run_tool("merge", session, "--write")
|
||||
catalog.write_text(FIXTURE.replace("こんにちは</translation>", "やあ</translation>"), encoding="utf-8")
|
||||
self.run_tool("apply", session, "--write")
|
||||
self.assertIn("<translation>やあ</translation>", catalog.read_text(encoding="utf-8"))
|
||||
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog, session, manifest = self.start_session(root)
|
||||
self.complete_responses(session, manifest)
|
||||
self.run_tool("merge", session, "--write")
|
||||
catalog.write_text(FIXTURE.replace("レジスタ", "外部変更"), encoding="utf-8")
|
||||
changed = catalog.read_bytes()
|
||||
result = self.run_tool("apply", session, "--write", expect=1)
|
||||
self.assertIn("changed outside this session", result.stdout)
|
||||
self.assertEqual(changed, catalog.read_bytes())
|
||||
|
||||
def test_empty_active_and_missing_translation_can_be_repaired(self) -> None:
|
||||
fixture = FIXTURE.replace(
|
||||
"</context>",
|
||||
" <message>\n"
|
||||
" <source>Empty active</source>\n"
|
||||
" <translation></translation>\n"
|
||||
" </message>\n"
|
||||
" <message>\n"
|
||||
" <source>Missing active</source>\n"
|
||||
" </message>\n"
|
||||
"</context>",
|
||||
1,
|
||||
)
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
catalog, session, manifest = self.start_session(
|
||||
root, fixture, "--include-empty-active"
|
||||
)
|
||||
found = {}
|
||||
for task in manifest["tasks"]:
|
||||
response = self.response_path(session, task)
|
||||
decisions = []
|
||||
for record in self.task_records(session, task):
|
||||
if record["source"] in {"Empty active", "Missing active"}:
|
||||
found[record["source"]] = True
|
||||
decisions.append({"id": record["id"], "translation": "修復済み"})
|
||||
else:
|
||||
decisions.append(self.decision_for(record))
|
||||
self.append_decisions(response, decisions)
|
||||
self.assertEqual({"Empty active", "Missing active"}, set(found))
|
||||
self.run_tool("merge", session, "--write")
|
||||
self.run_tool("apply", session, "--write")
|
||||
self.assertEqual(2, catalog.read_text(encoding="utf-8").count("<translation>修復済み</translation>"))
|
||||
|
||||
def test_validation_reports_lines_and_supports_strict_required_tags(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
catalog = Path(directory) / "catalog.ts"
|
||||
broken = FIXTURE.replace(
|
||||
@@ -265,29 +536,18 @@ class TranslationToolTests(unittest.TestCase):
|
||||
1,
|
||||
)
|
||||
catalog.write_text(broken, encoding="utf-8")
|
||||
translation_offset = broken.index("<translation>プレースホルダーなし</translation>")
|
||||
expected_line = broken.count("\n", 0, translation_offset) + 1
|
||||
result = self.run_tool("validate_ts.py", catalog, "--placeholders-only", expect=1)
|
||||
self.assertIn(f"{catalog}:{expected_line}:", result.stdout)
|
||||
result = self.run_validator(catalog, "--placeholders-only", expect=1)
|
||||
self.assertIn("[source: alpha.cpp:1]", result.stdout)
|
||||
|
||||
def test_validation_warns_for_placeholder_count_mismatch(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
catalog = Path(directory) / "catalog.ts"
|
||||
catalog.write_text(
|
||||
FIXTURE.replace(
|
||||
"Hello %1 {0} ${title} %.1f <strong>world</strong>",
|
||||
"Use {0}, then use {0} again",
|
||||
).replace(
|
||||
'<translation type="unfinished"></translation>',
|
||||
"<translation>Usar {0}</translation>",
|
||||
1,
|
||||
),
|
||||
encoding="utf-8",
|
||||
missing_tag = FIXTURE.replace(
|
||||
'<translation type="unfinished"></translation>',
|
||||
"<translation>Hello %1 {0} ${title} %.1f world</translation>",
|
||||
1,
|
||||
)
|
||||
result = self.run_tool("validate_ts.py", catalog, "--placeholders-only")
|
||||
self.assertIn("WARNING:", result.stdout)
|
||||
self.assertIn("placeholder count mismatch", result.stdout)
|
||||
catalog.write_text(missing_tag, encoding="utf-8")
|
||||
self.run_validator(catalog)
|
||||
strict = self.run_validator(catalog, "--strict-required-tags", expect=1)
|
||||
self.assertIn("missing rich-text tags", strict.stdout)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -7,7 +7,9 @@ import bisect
|
||||
import collections
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import tempfile
|
||||
import xml.etree.ElementTree as ET
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
@@ -15,7 +17,6 @@ from typing import Iterable, Iterator, Sequence
|
||||
|
||||
|
||||
SKIPPED_TYPES = frozenset({"vanished", "obsolete"})
|
||||
SCHEMA = "duckstation-qt-translation-batch-v1"
|
||||
|
||||
QT_PLACEHOLDER_RE = re.compile(r"%(?:L?\d+|Ln|n)")
|
||||
PRINTF_PLACEHOLDER_RE = re.compile(r"%(?!%)(?:(?:[-+0#]+\d*|\d+)?(?:\.\d+)?[diuoxXfFeEgGaAcsp])")
|
||||
@@ -296,7 +297,7 @@ def missing_rich_tags(source: str, translation: str) -> collections.Counter[str]
|
||||
return extract_rich_tags(source) - extract_rich_tags(translation)
|
||||
|
||||
|
||||
def message_to_batch_record(message: CatalogMessage) -> dict[str, object]:
|
||||
def message_to_task_record(message: CatalogMessage) -> dict[str, object]:
|
||||
return {
|
||||
"record_type": "message",
|
||||
"id": message.identifier,
|
||||
@@ -305,9 +306,7 @@ def message_to_batch_record(message: CatalogMessage) -> dict[str, object]:
|
||||
"current_type": message.translation_type,
|
||||
"current_translation": message.translation,
|
||||
"current_plural_translations": message.plural_translations,
|
||||
"accept_current": False,
|
||||
"target_translation": None,
|
||||
"target_plural_translations": None,
|
||||
"plural_arity": len(message.plural_translations),
|
||||
"suggestions": [],
|
||||
}
|
||||
|
||||
@@ -334,12 +333,36 @@ def load_jsonl(paths: Sequence[Path | str]) -> tuple[list[dict[str, object]], li
|
||||
return metadata, records
|
||||
|
||||
|
||||
def write_jsonl(path: Path, metadata: dict[str, object], records: Sequence[dict[str, object]]) -> None:
|
||||
def atomic_write_text(path: Path, text: str, mode: int | None = None) -> None:
|
||||
"""Write text beside its destination and atomically replace the destination."""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as stream:
|
||||
stream.write(json.dumps(metadata, ensure_ascii=False, sort_keys=True) + "\n")
|
||||
for record in records:
|
||||
stream.write(json.dumps(record, ensure_ascii=False, sort_keys=True) + "\n")
|
||||
if mode is None and path.exists():
|
||||
mode = path.stat().st_mode
|
||||
temporary_name: str | None = None
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(
|
||||
"w", encoding="utf-8", newline="", dir=path.parent, delete=False
|
||||
) as stream:
|
||||
temporary_name = stream.name
|
||||
stream.write(text)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
if mode is not None:
|
||||
os.chmod(temporary_name, mode)
|
||||
os.replace(temporary_name, path)
|
||||
finally:
|
||||
if temporary_name and os.path.exists(temporary_name):
|
||||
os.unlink(temporary_name)
|
||||
|
||||
|
||||
def jsonl_text(metadata: dict[str, object], records: Sequence[dict[str, object]]) -> str:
|
||||
lines = [json.dumps(metadata, ensure_ascii=False, sort_keys=True)]
|
||||
lines.extend(json.dumps(record, ensure_ascii=False, sort_keys=True) for record in records)
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def text_sha256(text: str) -> str:
|
||||
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def scan_raw_messages(text: str) -> dict[str, RawMessageSpan]:
|
||||
@@ -392,10 +415,17 @@ def remove_unfinished_attribute(attributes: str) -> str:
|
||||
def replace_translation_node(block: str, singular: str | None, plurals: Sequence[str] | None) -> str:
|
||||
match = TRANSLATION_RE.search(block)
|
||||
if match is None:
|
||||
raise ValueError("message has no translation element")
|
||||
attributes = remove_unfinished_attribute(match.group("attrs") or match.group("self_attrs") or "")
|
||||
line_start = block.rfind("\n", 0, match.start()) + 1
|
||||
indent = block[line_start:match.start()]
|
||||
closing = block.rfind("</message>")
|
||||
if closing < 0:
|
||||
raise ValueError("message has no closing element")
|
||||
line_start = block.rfind("\n", 0, closing) + 1
|
||||
message_indent = block[line_start:closing]
|
||||
indent = f"{message_indent} "
|
||||
attributes = ""
|
||||
else:
|
||||
attributes = remove_unfinished_attribute(match.group("attrs") or match.group("self_attrs") or "")
|
||||
line_start = block.rfind("\n", 0, match.start()) + 1
|
||||
indent = block[line_start:match.start()]
|
||||
if plurals is not None:
|
||||
forms = "\n".join(f"{indent} <numerusform>{xml_escape_text(value)}</numerusform>" for value in plurals)
|
||||
replacement = f"<translation{attributes}>\n{forms}\n{indent}</translation>"
|
||||
@@ -403,4 +433,6 @@ def replace_translation_node(block: str, singular: str | None, plurals: Sequence
|
||||
if singular is None:
|
||||
raise ValueError("missing singular target translation")
|
||||
replacement = f"<translation{attributes}>{xml_escape_text(singular)}</translation>"
|
||||
if match is None:
|
||||
return f"{block[:line_start]}{indent}{replacement}\n{block[line_start:]}"
|
||||
return block[: match.start()] + replacement + block[match.end() :]
|
||||
|
||||
@@ -38,6 +38,11 @@ def parse_args() -> argparse.Namespace:
|
||||
action="store_true",
|
||||
help="check placeholders only, allowing incomplete translations",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--strict-required-tags",
|
||||
action="store_true",
|
||||
help="treat missing or unbalanced rich-text tags as errors",
|
||||
)
|
||||
parser.add_argument("--strict-extra-tags", action="store_true", help="treat added rich-text tags as errors")
|
||||
return parser.parse_args()
|
||||
|
||||
@@ -129,11 +134,11 @@ def main() -> int:
|
||||
output.append(f"{label}: additional rich-text tags: {dict(extras)}")
|
||||
extras = missing_rich_tags(message.identity.source, value)
|
||||
if extras:
|
||||
output = errors if args.strict_extra_tags else warnings
|
||||
output = errors if args.strict_required_tags or args.strict_extra_tags else warnings
|
||||
output.append(f"{label}: missing rich-text tags: {dict(extras)}")
|
||||
unbalanced = unbalanced_rich_tags(value)
|
||||
if unbalanced:
|
||||
output = errors if args.strict_extra_tags else warnings
|
||||
output = errors if args.strict_required_tags or args.strict_extra_tags else warnings
|
||||
output.append(f"{label}: unbalanced rich-text tags (opening, closing): {unbalanced}")
|
||||
if (
|
||||
message.identity.numerus
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
|
||||
TRANSLATION_LIST_ENTRY("English", "en", "en-US", QLocale::English, QLocale::UnitedStates)
|
||||
TRANSLATION_LIST_ENTRY("Azərbaycanca", "az", "az-AZ", QLocale::Azerbaijani, QLocale::Azerbaijan)
|
||||
TRANSLATION_LIST_ENTRY("Deutsch", "de", "de-DE", QLocale::German, QLocale::Germany)
|
||||
TRANSLATION_LIST_ENTRY("Español de Hispanoamérica", "es", "es-ES", QLocale::Spanish, QLocale::LatinAmerica)
|
||||
TRANSLATION_LIST_ENTRY("Español de España", "es-ES", "es-ES", QLocale::Spanish, QLocale::Spain)
|
||||
TRANSLATION_LIST_ENTRY("Français", "fr", "fr-FR", QLocale::French, QLocale::France)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -782,14 +782,6 @@ Unread messages: {}</source>
|
||||
<numerusform>実績バッジを先読みしています (残り %n)...</numerusform>
|
||||
</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Failed to read executable from disc.</source>
|
||||
<translation>ディスクから実行ファイルを読み取れませんでした。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Achievements have been disabled.</source>
|
||||
<translation>実績が無効になりました。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>{0}, {1}.</source>
|
||||
<translation>{0}、{1}。</translation>
|
||||
@@ -2174,6 +2166,18 @@ WAV files must be stereo and use a sample rate of 44100hz.</source>
|
||||
<translation>{0} のサンプルレートは {1} Hz、チャンネル数は {2} です。
|
||||
WAV ファイルはステレオかつサンプルレート 44100 Hz である必要があります。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Preload Image To RAM</source>
|
||||
<translation>RAM にイメージを先読みする</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Allocating {} MB memory for precaching...</source>
|
||||
<translation>先読み用に {} MB のメモリを確保しています...</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Loading Track {0} ({1})...</source>
|
||||
<translation>トラック {0}({1})を読み込んでいます...</translation>
|
||||
</message>
|
||||
</context>
|
||||
<context>
|
||||
<name>CDImageHasher</name>
|
||||
@@ -2245,14 +2249,6 @@ Your dump may be corrupted, or the physical disc is scratched.</source>
|
||||
<source>Container:</source>
|
||||
<translation>コンテナー:</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Capture Video</source>
|
||||
<translation>映像をキャプチャ</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Capture Audio</source>
|
||||
<translation>音声をキャプチャ</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Codec:</source>
|
||||
<translation>コーデック:</translation>
|
||||
@@ -2351,21 +2347,21 @@ Your dump may be corrupted, or the physical disc is scratched.</source>
|
||||
<translation>コンテナー</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Determines the file format used to contain the captured audio/video.</source>
|
||||
<translation>キャプチャした音声/映像を格納するファイル形式を指定します。</translation>
|
||||
<source>Determines the file format used to contain the captured video.</source>
|
||||
<translation>キャプチャした動画を保存するファイル形式を指定します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Determines the file format used to contain the captured audio.</source>
|
||||
<translation>キャプチャした音声を保存するファイル形式を指定します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>%1 (Unknown)</source>
|
||||
<translation>%1(不明)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Specifies the directory where media capture (video/audio) will be saved.</source>
|
||||
<translation>メディアキャプチャ (映像/音声) を保存するディレクトリを指定します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Checked</source>
|
||||
<translation>オン</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Captures video to the chosen file when media capture is started. If unchecked, the file will only contain audio.</source>
|
||||
<translation>メディアキャプチャの開始時に、選択したファイルへ映像をキャプチャします。オフの場合、ファイルには音声のみが含まれます。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Video Codec</source>
|
||||
<translation>映像コーデック</translation>
|
||||
@@ -2418,10 +2414,6 @@ Your dump may be corrupted, or the physical disc is scratched.</source>
|
||||
<source>Parameters passed to the selected video codec.<br><b>You must use '=' to separate key from value and ':' to separate two pairs from each other.</b><br>For example: "crf = 21 : preset = veryfast"</source>
|
||||
<translation>選択した映像コーデックに渡すパラメーターです。<br><b>キーと値の区切りには「=」を、各ペアの区切りには「:」を使用する必要があります。</b><br>例: "crf = 21 : preset = veryfast"</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Captures audio to the chosen file when media capture is started. If unchecked, the file will only contain video.</source>
|
||||
<translation>メディアキャプチャの開始時に、選択したファイルへ音声をキャプチャします。オフの場合、ファイルには映像のみが含まれます。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Audio Codec</source>
|
||||
<translation>音声コーデック</translation>
|
||||
@@ -9082,10 +9074,6 @@ Do you want to {0} anyway?</source>
|
||||
<source>Enable Cheats</source>
|
||||
<translation>チートを有効にする</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Search...</source>
|
||||
<translation>検索...</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Sort Alphabetically</source>
|
||||
<translation>アルファベット順に並べ替え</translation>
|
||||
@@ -10004,10 +9992,6 @@ Scanning recursively takes more time, but will identify files in subdirectories.
|
||||
<source>All Regions</source>
|
||||
<translation>すべてのリージョン</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Search...</source>
|
||||
<translation>検索...</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>.cue (Cue Sheets)
|
||||
.iso (Single Track Image)
|
||||
@@ -11216,13 +11200,6 @@ Scanning recursively takes more time, but will identify files in subdirectories.
|
||||
<translation>ライトガンの水平位置に適用するオフセットです。</translation>
|
||||
</message>
|
||||
</context>
|
||||
<context>
|
||||
<name>HotkeySettingsWidget</name>
|
||||
<message>
|
||||
<source>Search...</source>
|
||||
<translation>検索...</translation>
|
||||
</message>
|
||||
</context>
|
||||
<context>
|
||||
<name>Hotkeys</name>
|
||||
<message>
|
||||
@@ -11449,6 +11426,14 @@ Scanning recursively takes more time, but will identify files in subdirectories.
|
||||
<source>Toggle Media Capture</source>
|
||||
<translation>メディアキャプチャを切り替え</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Toggle Audio Capture</source>
|
||||
<translation>オーディオキャプチャの切り替え</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Toggle Video Capture</source>
|
||||
<translation>ビデオキャプチャの切り替え</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Rotate Display Clockwise</source>
|
||||
<translation>表示を時計回りに回転</translation>
|
||||
@@ -12668,6 +12653,10 @@ Shift+クリックで複数のバインドを設定します。</translation>
|
||||
<source>Sort B&y</source>
|
||||
<translation>並べ替え(&Y)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Allows you to record audio and/or video from the content.</source>
|
||||
<translation>コンテンツの音声や映像を録画できます。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Start &File...</source>
|
||||
<translation>イメージ起動(&F)...</translation>
|
||||
@@ -13048,6 +13037,34 @@ Shift+クリックで複数のバインドを設定します。</translation>
|
||||
<source>System Log</source>
|
||||
<translation>システムログ</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Start &Capture</source>
|
||||
<translation>キャプチャを開始(&C)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Starts recording audio and video.</source>
|
||||
<translation>音声と映像の録画を開始します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Stop &Capture</source>
|
||||
<translation>キャプチャを停止(&C)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Start &Video-Only Capture</source>
|
||||
<translation>動画のみのキャプチャを開始(&V)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Starts a video-only recording.</source>
|
||||
<translation>動画のみの録画を開始します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Start &Audio-Only Capture</source>
|
||||
<translation>音声のみのキャプチャを開始(&A)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Starts an audio-only recording.</source>
|
||||
<translation>音声のみの録音を開始します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Automatic</source>
|
||||
<translation>自動</translation>
|
||||
@@ -13640,6 +13657,10 @@ Do you want to delete the save state and boot the game anyway?</source>
|
||||
<source>RA: Updated achievement progress database.</source>
|
||||
<translation>RA: 実績進捗データベースを更新しました。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>No containers are available for the current backend.</source>
|
||||
<translation>現在のバックエンドで使用できるコンテナがありません。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>%1 Files (*.%2)</source>
|
||||
<translation>%1 ファイル (*.%2)</translation>
|
||||
@@ -13840,26 +13861,6 @@ This action cannot be undone.</source>
|
||||
<comment>VideoCodec</comment>
|
||||
<translation>HEVC (ハードウェアエンコード)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>VP9 with Software Encoding</source>
|
||||
<comment>VideoCodec</comment>
|
||||
<translation>VP9 (ソフトウェアエンコード)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>VP9 with Hardware Encoding</source>
|
||||
<comment>VideoCodec</comment>
|
||||
<translation>VP9 (ハードウェアエンコード)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>AV1 with Software Encoding</source>
|
||||
<comment>VideoCodec</comment>
|
||||
<translation>AV1 (ソフトウェアエンコード)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>AV1 with Hardware Encoding</source>
|
||||
<comment>VideoCodec</comment>
|
||||
<translation>AV1 (ハードウェアエンコード)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Advanced Audio Coding</source>
|
||||
<comment>AudioCodec</comment>
|
||||
@@ -13919,6 +13920,10 @@ FFmpeg は {} からダウンロードできます。
|
||||
libswresample: {}
|
||||
</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Audio codec '{0}' does not support {1} Hz samples, using {2} Hz.</source>
|
||||
<translation>オーディオコーデック「{0}」は {1} Hz のサンプルに対応していないため、{2} Hz を使用します。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Matroska Media Container</source>
|
||||
<comment>ContainerFormat</comment>
|
||||
@@ -14120,14 +14125,6 @@ Error: {1}</source>
|
||||
<source>Export File</source>
|
||||
<translation>データのエクスポート</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source><<</source>
|
||||
<translation><<コピー</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>>></source>
|
||||
<translation>コピー>></translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>New Card...</source>
|
||||
<translation>新規作成...</translation>
|
||||
@@ -14180,6 +14177,18 @@ Error: {1}</source>
|
||||
<source> (Deleted)</source>
|
||||
<translation> (削 除)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>PNG Images (*.png);;JPEG Images (*.jpg *.jpeg);;WebP Images (*.webp)</source>
|
||||
<translation>PNG 画像(*.png);;JPEG 画像(*.jpg *.jpeg);;WebP 画像(*.webp)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Animated PNG Images (*.png)</source>
|
||||
<translation>アニメーション PNG 画像(*.png)</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Extract Icon</source>
|
||||
<translation>アイコンを抽出</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Select Memory Card</source>
|
||||
<translation>メモリーカードを選択</translation>
|
||||
@@ -14201,6 +14210,22 @@ Error: {1}</source>
|
||||
<source>Insufficient blocks, this file needs %1 but only %2 are available.</source>
|
||||
<translation>空ブロックが不十分です、このファイルには %1 が必要ですが、使用できるのは %2 のみです。</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Extract Animated Icon</source>
|
||||
<translation>アニメーションアイコンを抽出</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Failed to extract icon from save file %1:
|
||||
%2</source>
|
||||
<translation>セーブファイル %1 からアイコンを抽出できませんでした:
|
||||
%2</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Failed to extract animated icon from save file %1:
|
||||
%2</source>
|
||||
<translation>セーブファイル %1 からアニメーションアイコンを抽出できませんでした:
|
||||
%2</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Failed to import memory card from %1:
|
||||
%2</source>
|
||||
@@ -15732,6 +15757,10 @@ Error: {1}</source>
|
||||
<source>Selected Preset:</source>
|
||||
<translation>選択中のプリセット:</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Refresh Overlay List</source>
|
||||
<translation>オーバーレイリストを更新</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Custom Configuration</source>
|
||||
<translation>カスタム設定</translation>
|
||||
@@ -15847,10 +15876,6 @@ Error: {1}</source>
|
||||
<source>Select Shader</source>
|
||||
<translation>シェーダーを選択</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>Search...</source>
|
||||
<translation>検索...</translation>
|
||||
</message>
|
||||
<message>
|
||||
<source>All</source>
|
||||
<translation>すべて</translation>
|
||||
@@ -16430,6 +16455,13 @@ Would you like to update the shortcut to point to the current location?</source>
|
||||
<translation>セーブスロット {0} を選択しました。</translation>
|
||||
</message>
|
||||
</context>
|
||||
<context>
|
||||
<name>SearchBox</name>
|
||||
<message>
|
||||
<source>Search...</source>
|
||||
<translation>検索...</translation>
|
||||
</message>
|
||||
</context>
|
||||
<context>
|
||||
<name>SelectDiscDialog</name>
|
||||
<message>
|
||||
|
||||
Reference in New Issue
Block a user