157 lines
5.9 KiB
Python
157 lines
5.9 KiB
Python
#!/usr/bin/env python3
|
|
|
|
|
|
"""Build a reviewable TSV inventory of likely French first-party strings."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import subprocess
|
|
from collections import Counter
|
|
from pathlib import Path
|
|
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
OUTPUT = ROOT / "reports" / "french_strings.tsv"
|
|
UNIQUE_OUTPUT = ROOT / "reports" / "french_strings_unique.tsv"
|
|
RUNTIME_OUTPUT = ROOT / "reports" / "french_runtime_strings_unique.tsv"
|
|
FIRST_PARTY = ("src/", "sql/", "sylvania/", "chantiers/", "docker/", "contrib/")
|
|
TEXT_SUFFIXES = {
|
|
".c", ".cc", ".cpp", ".cxx", ".h", ".hpp", ".inl",
|
|
".sql", ".md", ".txt", ".conf", ".dist", ".sh", ".bat",
|
|
}
|
|
|
|
ACCENTS = re.compile(r"[àâçéèêëîïôùûüÿœæÀÂÇÉÈÊËÎÏÔÙÛÜŸŒÆ]")
|
|
WORDS = re.compile(
|
|
r"\b(?:alors|aucun|avec|avoir|besoin|cette|ceux|comme|comment|contre|"
|
|
r"dans|depuis|désormais|doit|donc|encore|entre|êtes|faire|faut|"
|
|
r"jusqu|leur|leurs|mais|merci|notre|pour|pourquoi|quand|sans|sera|"
|
|
r"sont|tout|tous|très|une|votre|vous|voici)\b",
|
|
re.IGNORECASE,
|
|
)
|
|
CONTRACTIONS = re.compile(r"\b(?:c|d|j|l|m|n|qu|s|t)'", re.IGNORECASE)
|
|
DOUBLE_QUOTED = re.compile(r'"(?:\\.|[^"\\])*"')
|
|
SINGLE_QUOTED = re.compile(r"'(?:\\.|''|[^'\\])*'")
|
|
|
|
|
|
def tracked_files() -> list[Path]:
|
|
output = subprocess.check_output(
|
|
["git", "ls-files", "-z"], cwd=ROOT
|
|
).decode("utf-8", "surrogateescape")
|
|
result = []
|
|
for name in output.split("\0"):
|
|
if not name or not name.startswith(FIRST_PARTY):
|
|
continue
|
|
path = ROOT / name
|
|
if path.suffix.lower() in TEXT_SUFFIXES:
|
|
result.append(path)
|
|
return result
|
|
|
|
|
|
def likely_french(text: str, path: Path) -> bool:
|
|
if "�" in text or len(text.strip()) < 2:
|
|
return False
|
|
accents = len(ACCENTS.findall(text))
|
|
words = len(WORDS.findall(text))
|
|
contractions = len(CONTRACTIONS.findall(text))
|
|
french_locale_file = bool(
|
|
re.search(r"(?:frfr|_fr(?:\.|_)|french)", path.as_posix(), re.IGNORECASE)
|
|
)
|
|
if accents:
|
|
return True
|
|
if words >= 2 or (words >= 1 and contractions >= 1):
|
|
return True
|
|
return french_locale_file and (" " in text or len(text) >= 8)
|
|
|
|
|
|
def clean_literal(value: str) -> str:
|
|
value = value[1:-1]
|
|
return value.replace("\t", "\\t").replace("\r", "\\r").replace("\n", "\\n")
|
|
|
|
|
|
def main() -> None:
|
|
rows: list[tuple[str, int, str, str]] = []
|
|
seen: set[tuple[str, int, str]] = set()
|
|
|
|
for path in tracked_files():
|
|
relative = path.relative_to(ROOT).as_posix()
|
|
try:
|
|
lines = path.read_text(encoding="utf-8").splitlines()
|
|
except (UnicodeDecodeError, OSError):
|
|
continue
|
|
|
|
for line_number, line in enumerate(lines, 1):
|
|
patterns = [DOUBLE_QUOTED]
|
|
if path.suffix.lower() == ".sql":
|
|
patterns.append(SINGLE_QUOTED)
|
|
|
|
found_literal = False
|
|
for pattern in patterns:
|
|
for match in pattern.finditer(line):
|
|
text = clean_literal(match.group(0)).strip()
|
|
if likely_french(text, path):
|
|
key = (relative, line_number, text)
|
|
if key not in seen:
|
|
rows.append((relative, line_number, "literal", text))
|
|
seen.add(key)
|
|
found_literal = True
|
|
|
|
# Documentation and comments may not use quoted literals.
|
|
stripped = line.strip()
|
|
is_comment_or_doc = (
|
|
path.suffix.lower() in {".md", ".txt"}
|
|
or stripped.startswith(("//", "#", "--", "/*", "*"))
|
|
)
|
|
if is_comment_or_doc and not found_literal and likely_french(stripped, path):
|
|
text = stripped.replace("\t", "\\t")
|
|
key = (relative, line_number, text)
|
|
if key not in seen:
|
|
rows.append((relative, line_number, "comment/doc", text))
|
|
seen.add(key)
|
|
|
|
rows.sort(key=lambda row: (row[0], row[1], row[2], row[3]))
|
|
OUTPUT.parent.mkdir(parents=True, exist_ok=True)
|
|
with OUTPUT.open("w", encoding="utf-8", newline="\n") as report:
|
|
report.write("file\tline\tkind\tfrench_text\n")
|
|
for relative, line_number, kind, text in rows:
|
|
report.write(f"{relative}\t{line_number}\t{kind}\t{text}\n")
|
|
|
|
occurrences: dict[tuple[str, str], list[str]] = {}
|
|
for relative, line_number, kind, text in rows:
|
|
occurrences.setdefault((kind, text), []).append(f"{relative}:{line_number}")
|
|
with UNIQUE_OUTPUT.open("w", encoding="utf-8", newline="\n") as report:
|
|
report.write("kind\toccurrences\tfrench_text\tlocations\n")
|
|
for (kind, text), locations in sorted(
|
|
occurrences.items(), key=lambda item: (item[0][0], item[0][1])
|
|
):
|
|
report.write(
|
|
f"{kind}\t{len(locations)}\t{text}\t{' | '.join(locations)}\n"
|
|
)
|
|
|
|
runtime_occurrences: dict[str, list[str]] = {}
|
|
for relative, line_number, kind, text in rows:
|
|
if (
|
|
kind == "literal"
|
|
and relative.startswith(("src/", "sql/"))
|
|
and not relative.startswith("sql/test/")
|
|
):
|
|
runtime_occurrences.setdefault(text, []).append(f"{relative}:{line_number}")
|
|
with RUNTIME_OUTPUT.open("w", encoding="utf-8", newline="\n") as report:
|
|
report.write("occurrences\tfrench_text\trussian_text\tlocations\n")
|
|
for text, locations in sorted(runtime_occurrences.items()):
|
|
report.write(
|
|
f"{len(locations)}\t{text}\t\t{' | '.join(locations)}\n"
|
|
)
|
|
|
|
per_kind = Counter(row[2] for row in rows)
|
|
print(f"Report: {OUTPUT.relative_to(ROOT)}")
|
|
print(f"Total: {len(rows)}")
|
|
print(f"Unique: {len(occurrences)}")
|
|
print(f"Unique runtime literals: {len(runtime_occurrences)}")
|
|
for kind, count in sorted(per_kind.items()):
|
|
print(f"{kind}: {count}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|