diff --git a/docs/reference/core.md b/docs/reference/core.md index dc9255e697..d94bc04621 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -65,20 +65,28 @@ specify init my-project --integration copilot --preset compliance ## Naming Features with the Helper Scripts When calling the bundled `create-new-feature` helper scripts directly, generated -names retain only ASCII letters and digits. A description entirely in a non-Latin -script, or made only of punctuation, can therefore produce an empty suffix such -as `001-`. The scripts warn on stderr when this happens, including during a dry -run; JSON output remains parseable. +names retain Unicode letters and decimal digits in UTF-8, so a description such as +`添加用户` produces `001-添加用户`. Descriptions made only of punctuation can still +produce an empty suffix such as `001-`; the scripts warn on stderr when this +happens, including during a dry run. JSON output remains parseable. -Keep the original description and supply a readable ASCII short name: +To choose a different name, keep the original description and supply a short name: ```bash -bash .specify/scripts/bash/create-new-feature.sh --json --short-name user-auth "添加用户" +bash .specify/scripts/bash/create-new-feature.sh --json --short-name 用户管理 "添加用户" ``` The Python helper also accepts `--short-name`; the PowerShell helper uses `-ShortName`. A supplied short name is cleaned by the same rules, so it must -contain at least one ASCII letter or digit. +contain at least one letter or digit. For non-ASCII names, the Bash helper needs +an installed UTF-8 locale and a Python 3 interpreter for Unicode classification. +ASCII input, including tabs and newlines, is sanitized without either requirement. +If `LC_ALL` is non-empty, Bash uses that locale rather than selecting another: +Unicode names fail with an error if the selected locale is not usable for UTF-8 +names. With `LC_ALL` unset or empty, Bash selects an installed UTF-8 locale even +when `LANG` or `LC_CTYPE` names a non-UTF-8 locale. +ASCII capitals are lowercased; non-ASCII letter casing is preserved across the +script variants. ## Check Installed Tools diff --git a/scripts/bash/common.sh b/scripts/bash/common.sh index 7b1516d29c..ff18cba4d9 100644 --- a/scripts/bash/common.sh +++ b/scripts/bash/common.sh @@ -112,7 +112,7 @@ read_feature_json_feature_directory() { fi if [[ -z "$_fd" ]] && command -v python3 >/dev/null 2>&1; then # Use Python so pretty-printed/multi-line JSON still parses correctly. - if ! _fd=$(python3 -c "import json,sys; d=json.load(open(sys.argv[1])); v=d.get('feature_directory'); print(v if v else '')" "$fj" 2>/dev/null); then + if ! _fd=$(python3 -c "import json,sys; d=json.load(open(sys.argv[1], encoding='utf-8')); v=d.get('feature_directory'); print(v if v else '')" "$fj" 2>/dev/null); then _fd='' fi fi diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index 1367014908..3ce2f9011c 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -146,17 +146,82 @@ spec_prefix_exists() { # Function to clean and format a branch name # -# Three details keep this byte-identical to the Python and PowerShell twins: -# * LC_ALL=C -- in a UTF-8 locale glibc resolves the a-z *range* through -# collation, so [^a-z0-9] keeps accented lowercase letters that -# re.sub(r"[^a-z0-9]", ...) and .NET's -replace both strip. +# Three details keep this consistent with the Python and PowerShell twins: +# * Unicode classification uses Python: POSIX [:alnum:] differs by platform. # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. +contains_non_ascii() { + LC_ALL=C grep -q '[^[:print:][:cntrl:]]' +} + +UNICODE_LOCALE="" +locale_candidates=(C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_CTYPE:-${LANG:-}}") +if [ -n "${LC_ALL:-}" ]; then + locale_candidates=("$LC_ALL") +fi +for candidate in "${locale_candidates[@]}"; do + if [ -n "$candidate" ] && [ "$(printf 'é。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é-' ]; then + UNICODE_LOCALE="$candidate" + break + fi +done + +if [ -z "$UNICODE_LOCALE" ]; then + UNICODE_LOCALE=C + if printf '%s' "${SHORT_NAME:-$FEATURE_DESCRIPTION}" | contains_non_ascii; then + if [ -n "${LC_ALL:-}" ]; then + echo "Error: A UTF-8 locale is required to create a Unicode feature name; LC_ALL=$LC_ALL is not usable" >&2 + else + echo "Error: A UTF-8 locale is required to create a Unicode feature name" >&2 + fi + exit 1 + fi +fi + +unicode_words() { + local name="${1//$'\n'/ }" + local separator="$2" + if printf '%s' "$name" | contains_non_ascii; then + local -a python_cmd=() + local override="${SPECKIT_PYTHON_EXECUTABLE:-${SPECKIT_PYTHON:-}}" + if [ -n "$override" ] && command -v "$override" >/dev/null 2>&1 && + "$override" -c 'import sys; raise SystemExit(sys.version_info.major != 3)' >/dev/null 2>&1; then + python_cmd=("$override") + else + local python_line + while IFS= read -r python_line; do + python_cmd+=("$python_line") + done < <(_python3_command) + fi + if [ "${#python_cmd[@]}" -eq 0 ]; then + echo "Error: Python 3 is required to create a Unicode feature name" >&2 + return 1 + fi + printf '%s' "$name" | "${python_cmd[@]}" -c ' +import sys +value = sys.stdin.buffer.read().decode("utf-8") +lower = str.maketrans("ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz") +result = "".join( + char if char.isalpha() or char.isdecimal() else sys.argv[1] + for char in value.translate(lower) +) +sys.stdout.buffer.write(result.encode("utf-8")) +' "$separator" + else + printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed "s/[^a-z0-9]/$separator/g" + fi +} + clean_branch_name() { local name="$1" - local -x LC_ALL=C - printf '%s\n' "$name" | tr '[:upper:]' '[:lower:]' | sed 's/[^a-z0-9]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' + local cleaned + cleaned=$(unicode_words "$name" '-') || return 1 + printf '%s\n' "$cleaned" | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' +} + +branch_byte_count() { + printf '%s' "$1" | wc -c | tr -d '[:space:]' } # Fit a feature prefix and suffix within GitHub's branch-name limit. @@ -165,11 +230,25 @@ fit_branch_name() { local branch_suffix="$2" local branch_name="${feature_num}-${branch_suffix}" - if [ ${#branch_name} -gt $MAX_BRANCH_LENGTH ]; then + if [ "$(branch_byte_count "$branch_name")" -gt "$MAX_BRANCH_LENGTH" ]; then local prefix_length=$(( ${#feature_num} + 1 )) local max_suffix_length=$((MAX_BRANCH_LENGTH - prefix_length)) local truncated_suffix - truncated_suffix=$(printf '%s' "$branch_suffix" | cut -c "1-$max_suffix_length" | sed 's/-$//') + local -x LC_ALL="$UNICODE_LOCALE" + local low=0 high=${#branch_suffix} mid + if (( high > max_suffix_length )); then + high=$max_suffix_length + fi + while (( low < high )); do + mid=$(((low + high + 1) / 2)) + if [ "$(branch_byte_count "${branch_suffix:0:$mid}")" -le "$max_suffix_length" ]; then + low=$mid + else + high=$((mid - 1)) + fi + done + truncated_suffix="${branch_suffix:0:$low}" + truncated_suffix="${truncated_suffix%-}" branch_name="${feature_num}-${truncated_suffix}" fi @@ -209,12 +288,10 @@ generate_branch_name() { # Common stop words to filter out local stop_words="^(i|a|an|the|to|for|of|in|on|at|by|with|from|is|are|was|were|be|been|being|have|has|had|do|does|did|will|would|should|could|can|may|might|must|shall|this|that|these|those|my|your|our|their|want|need|add|get|set)$" - # Convert to lowercase and split into words. LC_ALL=C for the same - # collation reason documented on clean_branch_name, and so the `grep -qw` - # acronym probe below uses ASCII word boundaries like the Python twin's - # (?= 3 OR are potential acronyms) - if ! echo "$word" | grep -qiE "$stop_words"; then - if [ ${#word} -ge 3 ]; then + # Retain non-ASCII words even when shorter than three characters. + if ! printf '%s\n' "$word" | LC_ALL=C grep -qE "$stop_words"; then + if [ ${#word} -ge 3 ] || printf '%s' "$word" | contains_non_ascii; then meaningful_words+=("$word") # Keep short words that appear as an uppercase acronym in the original. # Uppercase via tr and match with grep -w (both portable) rather than # bash's 4+ "^^" case expansion (breaks on macOS bash 3.2) and \b (non-POSIX). - elif printf '%s' "$description" | grep -qw -- "$(printf '%s' "$word" | tr '[:lower:]' '[:upper:]')"; then + elif printf '%s' "$description" | LC_ALL=C grep -qw -- "$(printf '%s' "$word" | LC_ALL=C tr '[:lower:]' '[:upper:]')"; then meaningful_words+=("$word") fi fi @@ -266,7 +343,7 @@ else fi if [ -z "$BRANCH_SUFFIX" ]; then - echo "[specify] Warning: Feature name is empty after removing unsupported characters. Use --short-name with ASCII letters or digits (for example, user-auth)." >&2 + echo "[specify] Warning: Feature name is empty after removing unsupported characters. Use --short-name with letters or digits (for example, user-auth)." >&2 fi # Warn if --number and --timestamp are both specified @@ -339,8 +416,8 @@ ORIGINAL_BRANCH_NAME="${FEATURE_NUM}-${BRANCH_SUFFIX}" BRANCH_NAME=$(fit_branch_name "$FEATURE_NUM" "$BRANCH_SUFFIX") if [ "$BRANCH_NAME" != "$ORIGINAL_BRANCH_NAME" ]; then >&2 echo "[specify] Warning: Branch name exceeded GitHub's 244-byte limit" - >&2 echo "[specify] Original: $ORIGINAL_BRANCH_NAME (${#ORIGINAL_BRANCH_NAME} bytes)" - >&2 echo "[specify] Truncated to: $BRANCH_NAME (${#BRANCH_NAME} bytes)" + >&2 echo "[specify] Original: $ORIGINAL_BRANCH_NAME ($(branch_byte_count "$ORIGINAL_BRANCH_NAME") bytes)" + >&2 echo "[specify] Truncated to: $BRANCH_NAME ($(branch_byte_count "$BRANCH_NAME") bytes)" fi FEATURE_DIR="$SPECS_DIR/$BRANCH_NAME" diff --git a/scripts/powershell/create-new-feature.ps1 b/scripts/powershell/create-new-feature.ps1 index c90bd26e46..cd8a2d2b3c 100644 --- a/scripts/powershell/create-new-feature.ps1 +++ b/scripts/powershell/create-new-feature.ps1 @@ -83,10 +83,32 @@ function Test-SpecPrefixInUse { Select-Object -First 1) } +function ConvertTo-AsciiLower { + param([string]$Name) + + return [regex]::Replace($Name, '[A-Z]', { param($match) $match.Value.ToLowerInvariant() }) +} + +function ConvertTo-UnicodeWords { + param([string]$Name, [string]$Separator) + + $lowerName = ConvertTo-AsciiLower -Name $Name + return [regex]::Replace($lowerName, '[\uD800-\uDBFF][\uDC00-\uDFFF]|[^\p{L}\p{Nd}]', { + param($match) + if ($match.Length -eq 2) { + $category = [System.Globalization.CharUnicodeInfo]::GetUnicodeCategory($match.Value, 0) + if ($category.ToString() -match '(Letter|DecimalDigitNumber)$') { + return $match.Value + } + } + return $Separator + }) +} + function ConvertTo-CleanBranchName { param([string]$Name) - return $Name.ToLower() -replace '[^a-z0-9]', '-' -replace '-{2,}', '-' -replace '^-', '' -replace '-$', '' + return (ConvertTo-UnicodeWords -Name $Name -Separator '-') -replace '-{2,}', '-' -replace '^-', '' -replace '-$', '' } function Get-FittedBranchName { @@ -96,10 +118,15 @@ function Get-FittedBranchName { ) $fittedName = "$FeatureNum-$BranchSuffix" - if ($fittedName.Length -gt $maxBranchLength) { + if ([System.Text.Encoding]::UTF8.GetByteCount($fittedName) -gt $maxBranchLength) { $prefixLength = $FeatureNum.Length + 1 $maxSuffixLength = $maxBranchLength - $prefixLength - $truncatedSuffix = $BranchSuffix.Substring(0, [Math]::Min($BranchSuffix.Length, $maxSuffixLength)) + $bytes = [System.Text.Encoding]::UTF8.GetBytes($BranchSuffix) + $bytesToUse = $maxSuffixLength + while ($bytesToUse -gt 0 -and ($bytes[$bytesToUse] -band 0xC0) -eq 0x80) { + $bytesToUse-- + } + $truncatedSuffix = [System.Text.Encoding]::UTF8.GetString($bytes, 0, $bytesToUse) $truncatedSuffix = $truncatedSuffix -replace '-$', '' $fittedName = "$FeatureNum-$truncatedSuffix" } @@ -132,18 +159,19 @@ function Get-BranchName { 'want', 'need', 'add', 'get', 'set' ) - # Convert to lowercase and extract words (alphanumeric only) - $cleanName = $Description.ToLower() -replace '[^a-z0-9\s]', ' ' + # Lowercase ASCII and extract Unicode words, matching the shell variant. + $cleanName = ConvertTo-UnicodeWords -Name $Description -Separator ' ' $words = $cleanName -split '\s+' | Where-Object { $_ } # Filter words: remove stop words and words shorter than 3 chars (unless they're uppercase acronyms in original) $meaningfulWords = @() foreach ($word in $words) { # Skip stop words - if ($stopWords -contains $word) { continue } + if ($stopWords -ccontains $word) { continue } - # Keep words that are length >= 3 OR appear as uppercase in original (likely acronyms) - if ($word.Length -ge 3) { + # Keep Unicode words even when short; ASCII words still need three + # characters or an uppercase acronym in the original. + if ($word.Length -ge 3 -or $word -match '[^\x00-\x7F]') { $meaningfulWords += $word } elseif ($Description -cmatch "(? str: _MAX_BRANCH_LENGTH = 244 _MAX_FEATURE_NUMBER = 2**63 - 1 +_ASCII_LOWER = str.maketrans( + "ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz" +) def _int64_from_digits(value: str) -> int | None: @@ -167,18 +170,25 @@ def _parse_args(argv: list[str], argv0: str) -> Args: def _clean_branch_name(name: str) -> str: - cleaned = re.sub(r"[^a-z0-9]", "-", name.lower()) + cleaned = _unicode_words(name, "-") cleaned = re.sub(r"-+", "-", cleaned) return cleaned.strip("-") +def _unicode_words(name: str, separator: str) -> str: + return "".join( + char if char.isalpha() or char.isdecimal() else separator + for char in name.translate(_ASCII_LOWER) + ) + + def _generate_branch_name(description: str) -> str: - clean = re.sub(r"[^a-z0-9]", " ", description.lower()) + clean = _unicode_words(description, " ") meaningful: list[str] = [] for word in clean.split(): if word in _STOP_WORDS: continue - if len(word) >= 3: + if len(word) >= 3 or not word.isascii(): meaningful.append(word) # Keep short words that appear as an uppercase acronym in the original, # mirroring the bash twin's case-sensitive `grep -qw` check. @@ -217,11 +227,13 @@ def _get_highest_from_specs(specs_dir: Path) -> int: def _fit_branch_name(feature_num: str, branch_suffix: str) -> str: """Fit a feature prefix and suffix within GitHub's branch-name limit.""" branch_name = f"{feature_num}-{branch_suffix}" - if len(branch_name) <= _MAX_BRANCH_LENGTH: + if len(branch_name.encode("utf-8")) <= _MAX_BRANCH_LENGTH: return branch_name max_suffix_length = _MAX_BRANCH_LENGTH - (len(feature_num) + 1) - truncated_suffix = re.sub(r"-$", "", branch_suffix[:max_suffix_length]) + truncated_suffix = branch_suffix.encode("utf-8")[:max_suffix_length].decode( + "utf-8", errors="ignore" + ).rstrip("-") return f"{feature_num}-{truncated_suffix}" @@ -268,7 +280,7 @@ def main(argv: list[str] | None = None) -> int: if not branch_suffix: print( "[specify] Warning: Feature name is empty after removing unsupported characters. " - "Use --short-name with ASCII letters or digits (for example, user-auth).", + "Use --short-name with letters or digits (for example, user-auth).", file=sys.stderr, ) @@ -363,11 +375,12 @@ def main(argv: list[str] | None = None) -> int: ) print( f"[specify] Original: {original_branch_name} " - f"({len(original_branch_name)} bytes)", + f"({len(original_branch_name.encode('utf-8'))} bytes)", file=sys.stderr, ) print( - f"[specify] Truncated to: {branch_name} ({len(branch_name)} bytes)", + f"[specify] Truncated to: {branch_name} " + f"({len(branch_name.encode('utf-8'))} bytes)", file=sys.stderr, ) @@ -448,4 +461,6 @@ def main(argv: list[str] | None = None) -> int: if __name__ == "__main__": + sys.stdout.reconfigure(encoding="utf-8") + sys.stderr.reconfigure(encoding="utf-8") raise SystemExit(main()) diff --git a/tests/parity_helpers.py b/tests/parity_helpers.py index bc6edc0aca..56bee0806b 100644 --- a/tests/parity_helpers.py +++ b/tests/parity_helpers.py @@ -208,6 +208,7 @@ def collation_range_locale() -> str | None: input="é\n", capture_output=True, text=True, + encoding="utf-8", check=False, env=env, ) @@ -235,6 +236,7 @@ def run( cwd=repo, capture_output=True, text=True, + encoding="utf-8", check=False, env=env if env is not None else clean_env(), timeout=timeout, diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 36acefb7bc..177bdea30d 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -2,7 +2,12 @@ from __future__ import annotations +import os import re +import shlex +import shutil +import subprocess +import sys from pathlib import Path import pytest @@ -12,6 +17,8 @@ from tests.conftest import requires_bash from tests.parity_helpers import ( HAS_POWERSHELL, + WINDOWS_POWERSHELL, + _bash_posix_path, bash_cmd, break_wrap_layer, clean_env, @@ -19,7 +26,9 @@ install_composition_stack, install_scripts, json_stdout, + make_python3_path_shim, make_repo, + make_yaml_less_venv, normalize_repo_paths, normalize_script_names, ps_cmd, @@ -72,11 +81,11 @@ def repo_pair(tmp_path: Path) -> tuple[Path, Path]: @pytest.mark.parametrize( "description,short_name,suffix,warns", [ - ("添加用户", None, "", True), - ("добавить", None, "", True), + ("添加用户", None, "添加用户", False), + ("добавить", None, "добавить", False), ("!!! ??? ***", None, "", True), ("添加用户", "user-auth", "user-auth", False), - ("Add users", "用户", "", True), + ("Add users", "用户", "用户", False), ("Add user authentication", None, "user-authentication", False), ], ) @@ -89,7 +98,7 @@ def test_empty_feature_name_warning( suffix: str, warns: bool, ) -> None: - """Report unusable names without changing JSON or feature creation (#4574).""" + """Preserve UTF-8 names and warn only when no name remains (#4574).""" powershell = variant == "powershell" args = ["-Json" if powershell else "--json"] if dry_run: @@ -107,10 +116,250 @@ def test_empty_feature_name_warning( assert result.stderr.count(warning) == int(warns) if warns: assert ("-ShortName" if powershell else "--short-name") in result.stderr - assert "ASCII letters or digits" in result.stderr + assert "letters or digits" in result.stderr assert (repo / "specs" / f"001-{suffix}" / "spec.md").exists() is not dry_run +@pytest.mark.parametrize("json_mode", [False, True], ids=["text", "json"]) +@pytest.mark.parametrize("dry_run", [False, True], ids=["create", "dry_run"]) +def test_python_outputs_unicode_when_default_encoding_is_cp1252( + repo: Path, json_mode: bool, dry_run: bool +) -> None: + env = clean_env() + env["PYTHONIOENCODING"] = "cp1252" + args = [] + if json_mode: + args.append("--json") + if dry_run: + args.append("--dry-run") + args.append("添加用户") + + result = run(py_cmd(repo, SCRIPT, *args), repo, env) + + assert result.returncode == 0, result.stderr + assert "001-添加用户" in result.stdout + if not dry_run: + assert "001-添加用户" in result.stderr + if json_mode: + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + else: + assert "BRANCH_NAME: 001-添加用户" in result.stdout + + +@pytest.mark.parametrize( + "variant", + [ + pytest.param("bash", marks=requires_bash), + "python", + pytest.param( + "powershell", + marks=pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available"), + ), + ], +) +@pytest.mark.parametrize( + ("short_name", "expected"), + [ + ("x²", "001-x"), + ("xⅫ", "001-x"), + ("x٥", "001-x٥"), + ("x𝟘", "001-x𝟘"), + ], + ids=["superscript_number", "roman_numeral", "decimal_digit", "supplementary_decimal"], +) +def test_unicode_number_categories_match( + repo: Path, variant: str, short_name: str, expected: str +) -> None: + powershell = variant == "powershell" + args = ( + ("-Json", "-DryRun", "-ShortName", short_name, "x") + if powershell + else ("--json", "--dry-run", "--short-name", short_name, "x") + ) + command = {"bash": bash_cmd, "python": py_cmd, "powershell": ps_cmd}[variant] + + result = run(command(repo, SCRIPT, *args), repo) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + + +@requires_bash +def test_bash_reports_missing_utf8_locale_for_unicode_only( + repo: Path, tmp_path: Path +) -> None: + real_sed = shutil.which("sed") + assert real_sed is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + sed_shim = shim_dir / "sed" + sed_shim.write_text( + "#!/bin/sh\n" + """if [ "$1" = 's/[^[:alnum:]]/-/g' ]; then exit 1; fi\n""" + f"exec {shlex.quote(real_sed)} \"$@\"\n", + encoding="utf-8", + ) + sed_shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}:{env['PATH']}" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + for args in (("添加用户",), ("--short-name", "用户", "Add users")): + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", *args), repo, env) + assert result.returncode == 1 + assert result.stdout == "" + assert "Error: A UTF-8 locale is required" in result.stderr + assert not (repo / "specs").exists() + + +@requires_bash +@pytest.mark.parametrize("locale_name", ["C", "POSIX"]) +def test_bash_respects_explicit_non_utf8_lc_all(repo: Path, locale_name: str) -> None: + env = clean_env() + env["LC_ALL"] = locale_name + env["LANG"] = "C.UTF-8" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + for args in (("添加用户",), ("--short-name", "用户", "Add users")): + unicode_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", *args), repo, env + ) + assert unicode_result.returncode == 1 + assert unicode_result.stdout == "" + assert "A UTF-8 locale is required" in unicode_result.stderr + assert "LC_ALL" in unicode_result.stderr + assert not (repo / "specs").exists() + + +@requires_bash +@pytest.mark.parametrize( + "short_name", ["foo\tbar", "foo\nbar"], ids=["tab", "newline"] +) +def test_bash_ascii_controls_need_no_utf8_locale_or_python( + repo: Path, tmp_path: Path, short_name: str +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + + for lc_all in ("C", None): + env = clean_env() + if lc_all is not None: + env["LC_ALL"] = lc_all + else: + env.pop("LC_ALL", None) + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + bash = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", short_name, "x"), + repo, + env, + ) + py = run( + py_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", short_name, "x"), + repo, + env, + ) + assert bash.returncode == py.returncode == 0, bash.stderr + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == "001-foo-bar" + + +@requires_bash +def test_bash_requires_python_only_for_unicode_names( + repo: Path, tmp_path: Path +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}:{env['PATH']}" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + unicode_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env + ) + assert unicode_result.returncode == 1 + assert unicode_result.stdout == "" + assert "Error: Python 3 is required to create a Unicode feature name" in unicode_result.stderr + + +@requires_bash +def test_bash_unicode_uses_configured_python_without_pyyaml_in_spaced_path( + repo: Path, tmp_path: Path +) -> None: + no_yaml_exe = make_yaml_less_venv(tmp_path / "tool env") + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["SPECKIT_PYTHON_EXECUTABLE"] = _bash_posix_path(no_yaml_exe) + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + + +@requires_bash +def test_bash_unicode_uses_py_launcher_with_separate_version_arg( + repo: Path, tmp_path: Path +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + launcher = shim_dir / "py" + launcher.write_text( + "#!/bin/sh\n" + '[ "$1" = "-3" ] || exit 1\n' + "shift\n" + f'exec {shlex.quote(_bash_posix_path(Path(sys.executable)))} "$@"\n', + encoding="utf-8", + newline="\n", + ) + launcher.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + + def _run_all_variants_allow_existing( repo: Path, *, number: str, short_name: str ): @@ -173,10 +422,8 @@ def deny_listing(_path: Path): "I want to add the new API rate limiting feature for users", "Fix UI for DB sync", "a to the of", - # An acronym touching an accented letter: bash probes with `grep -qw` - # under LC_ALL=C, where the accent is a word boundary, so the Python - # twin must use explicit ASCII lookarounds rather than a Unicode \b. "Fix \u00e9DB\u00e9 sync", + "Ajouter la réservation hôtelière", ], ids=[ "plain", @@ -184,6 +431,7 @@ def deny_listing(_path: Path): "acronyms", "all_stop_words_fallback", "acronym_next_to_non_ascii", + "accented_words", ], ) def test_python_branch_name_generation_matches_bash( @@ -198,23 +446,78 @@ def test_python_branch_name_generation_matches_bash( @requires_bash -@pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") -def test_all_variants_keep_acronym_next_to_non_ascii(repo: Path) -> None: - """An acronym touching an accented letter survives in all three twins. +@pytest.mark.parametrize( + ("description", "expected"), + [("ſet account", "001-ſet-account"), ("Set account", "001-account")], +) +def test_bash_stop_words_match_only_ascii_words( + repo: Path, description: str, expected: str +) -> None: + bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) + py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) - bash probes for acronyms with `grep -qw` under LC_ALL=C, where an accented - letter is a non-word byte and therefore a boundary. Python's \\b and .NET's - \\b are Unicode-aware and saw "\u00e9DB\u00e9" as a single word, dropping the - acronym; all three now spell the boundary out as ASCII. - """ - description = "Fix \u00e9DB\u00e9 sync" + assert bash.returncode == py.returncode == 0 + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == expected + if HAS_POWERSHELL: + ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) + assert ps.returncode == 0, ps.stderr + assert json_stdout(ps) == json_stdout(py) + + +@requires_bash +def test_bash_stop_words_ignore_locale_case_folding(repo: Path, tmp_path: Path) -> None: + real_grep = shutil.which("grep") + assert real_grep is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + grep_shim = shim_dir / "grep" + grep_cmd = shlex.quote(_bash_posix_path(Path(real_grep))) + grep_shim.write_text( + "#!/bin/sh\n" + 'if [ "$1" = "-qiE" ]; then\n' + " IFS= read -r word\n" + ' [ "$word" = "ſet" ] && exit 0\n' + f' printf "%s\\n" "$word" | {grep_cmd} "$@"\n' + " exit $?\n" + "fi\n" + f'exec {grep_cmd} "$@"\n', + encoding="utf-8", + newline="\n", + ) + grep_shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "ſet account"), repo, env) + py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", "ſet account"), repo, env) + + assert bash.returncode == py.returncode == 0 + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == "001-ſet-account" + + +@requires_bash +@pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") +@pytest.mark.parametrize( + ("description", "expected"), + [ + ("Fix éDBé sync", "001-fix-édbé-sync"), + ("É DB sync", "001-É-db-sync"), + ("é DB sync", "001-é-db-sync"), + ], +) +def test_all_variants_keep_acronym_next_to_non_ascii( + repo: Path, description: str, expected: str +) -> None: + """Unicode words and adjacent ASCII acronyms survive together.""" bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) assert bash.returncode == py.returncode == ps.returncode == 0 assert json_stdout(py) == json_stdout(bash) == json_stdout(ps) - assert json_stdout(ps)["BRANCH_NAME"] == "001-fix-db-sync" + assert json_stdout(ps)["BRANCH_NAME"] == expected @requires_bash @@ -695,6 +998,60 @@ def test_python_persists_relative_feature_json(repo: Path) -> None: assert feature_json == f'{{"feature_directory":"specs/{branch}"}}\n' +@requires_bash +def test_bash_reads_unicode_feature_state_without_jq_under_legacy_encoding( + repo: Path, tmp_path: Path +) -> None: + created = run(bash_cmd(repo, SCRIPT, "--json", "添加用户"), repo) + assert created.returncode == 0, created.stderr + assert json_stdout(created)["BRANCH_NAME"] == "001-添加用户" + assert (repo / "specs/001-添加用户/spec.md").is_file() + + shim_dir = make_python3_path_shim(tmp_path / "bin") + (shim_dir / "sitecustomize.py").write_text( + "import builtins\n" + "_open = builtins.open\n" + "def legacy_open(file, *args, **kwargs):\n" + " if str(file).endswith('feature.json') and 'encoding' not in kwargs:\n" + " kwargs['encoding'] = 'cp1252'\n" + " return _open(file, *args, **kwargs)\n" + "builtins.open = legacy_open\n", + encoding="utf-8", + ) + for name in ("jq", "grep"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + env["PYTHONPATH"] = str(shim_dir) + env["PYTHONIOENCODING"] = "utf-8" + common = repo / ".specify/scripts/bash/common.sh" + + resolved = run( + [ + "bash", + "-c", + ( + 'source "$1"; paths=$(get_feature_paths --no-persist) || exit 1; ' + 'printf -v expected "FEATURE_DIR=%q" "$3"; ' + '[[ "$paths" == *"$expected"* ]] || exit 1; ' + 'read_feature_json_feature_directory "$2"' + ), + "bash", + str(common), + str(repo), + str(repo / "specs/001-添加用户"), + ], + repo, + env, + ) + + assert resolved.returncode == 0, resolved.stderr + assert resolved.stdout == "specs/001-添加用户" + assert (repo / resolved.stdout / "spec.md").is_file() + + def test_persist_feature_json_avoids_platform_newline_translation( tmp_path: Path, monkeypatch ) -> None: @@ -1166,15 +1523,7 @@ def test_all_variants_corrected_prefix_skips_timestamp_collision(repo: Path) -> def test_bash_branch_name_ignores_locale_collation( repo: Path, description: str ) -> None: - """Branch naming must not depend on the caller's locale. - - ``clean_branch_name``/``generate_branch_name`` sanitize with - ``sed 's/[^a-z0-9]/-/g'``. Run under a collation-ordered locale that class - keeps accented lowercase letters, so bash produced - ``001-ajouter-réservation-hôtelière`` where the Python and PowerShell twins - produce ``001-ajouter-servation-teli``: the same description yielded a - different ``specs/`` directory on two machines that differ only in ``LANG``. - """ + """Branch naming preserves each description's Unicode words across locales.""" locale_name = collation_range_locale() if locale_name is None: pytest.skip("no locale with collation-ordered [a-z] ranges available") @@ -1189,12 +1538,13 @@ def test_bash_branch_name_ignores_locale_collation( assert py.returncode == bash.returncode == 0 assert json_stdout(py) == json_stdout(bash) branch = json_stdout(bash)["BRANCH_NAME"] - assert isinstance(branch, str) and branch.isascii(), branch + assert isinstance(branch, str) and branch == { + "Añadir autenticación de usuario": "001-añadir-autenticación-usuario", + "Prüfung für Benutzer anlegen": "001-prüfung-für-benutzer-anlegen", + "Ajouter la réservation hôtelière": "001-ajouter-réservation-hôtelière", + }[description] - # The run above reaches generate_branch_name. --short-name reaches - # clean_branch_name, a separate function carrying its own LC_ALL=C, so - # exercise the accented value through both: neither copy can then regress - # on its own without a failure here. + # Both the generated and explicit-name paths must retain the input's words. short_args = ("--json", "--dry-run", "--short-name", description, "x") bash_short = run(bash_cmd(repo, SCRIPT, *short_args), repo, env) py_short = run(py_cmd(repo, SCRIPT, *short_args), repo, env) @@ -1202,7 +1552,11 @@ def test_bash_branch_name_ignores_locale_collation( assert py_short.returncode == bash_short.returncode == 0 assert json_stdout(py_short) == json_stdout(bash_short) short_branch = json_stdout(bash_short)["BRANCH_NAME"] - assert isinstance(short_branch, str) and short_branch.isascii(), short_branch + assert isinstance(short_branch, str) and short_branch == { + "Añadir autenticación de usuario": "001-añadir-autenticación-de-usuario", + "Prüfung für Benutzer anlegen": "001-prüfung-für-benutzer-anlegen", + "Ajouter la réservation hôtelière": "001-ajouter-la-réservation-hôtelière", + }[description] @requires_bash @@ -1260,22 +1614,18 @@ def test_python_dash_prefixed_short_name_matches_bash( @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") @pytest.mark.parametrize( - "description", - ["!!! ??? ***", "добавить", "添加用户"], + ("description", "expected"), + [ + ("!!! ??? ***", "001-"), + ("добавить", "001-добавить"), + ("添加用户", "001-添加用户"), + ], ids=["punctuation_only", "cyrillic", "han"], ) -def test_powershell_survives_description_with_no_ascii_words( - tmp_path: Path, description: str +def test_powershell_preserves_unicode_description( + tmp_path: Path, description: str, expected: str ): - """A description with no [a-z0-9] characters must not crash the PS twin. - - ``ConvertTo-CleanBranchName`` blanks every non-ASCII character, so the - fallback pipeline yields nothing and ``[string]::Join`` received ``$null`` - — an ArgumentNullException, made terminating by - ``$ErrorActionPreference = 'Stop'``. The script died with a .NET stack - trace and exit 1 where the bash and Python twins both return an empty - suffix. This fires for any feature phrased in a non-Latin script. - """ + """Non-Latin descriptions are usable; punctuation alone still warns.""" repo = _setup_repo(tmp_path) ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) @@ -1283,22 +1633,39 @@ def test_powershell_survives_description_with_no_ascii_words( assert ps.returncode == 0, ps.stderr assert "ArgumentNullException" not in ps.stderr assert "Join" not in ps.stderr - assert json_stdout(ps)["BRANCH_NAME"] == "001-" + assert json_stdout(ps)["BRANCH_NAME"] == expected @requires_bash @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") -def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): - """All three twins agree on the branch name for such a description.""" - description = "добавить" +@pytest.mark.parametrize( + ("description", "expected"), + [ + ("добавить", "001-добавить"), + ("给倒推引擎加正推能力", "001-给倒推引擎加正推能力"), + ("客户邮件。审核队列", "001-客户邮件-审核队列"), + ("ПРИВЕТ", "001-ПРИВЕТ"), + ("𠀀𠀁。功能", "001-𠀀𠀁-功能"), + ("😀!!!", "001-"), + ], +) +def test_no_ascii_word_description_matches_across_twins( + tmp_path: Path, description: str, expected: str +): + """All three twins agree on UTF-8 feature names and separators.""" bash_repo = _setup_repo(tmp_path, "b") py_repo = _setup_repo(tmp_path, "p") ps_repo = _setup_repo(tmp_path, "s") - bash = run(bash_cmd(bash_repo, SCRIPT, "--json", "--dry-run", description), bash_repo) - py = run(py_cmd(py_repo, SCRIPT, "--json", "--dry-run", description), py_repo) - ps = run(ps_cmd(ps_repo, SCRIPT, "-Json", "-DryRun", description), ps_repo) + env = clean_env() + env.pop("LC_ALL", None) + env["LANG"] = "C" + bash = run( + bash_cmd(bash_repo, SCRIPT, "--json", "--dry-run", description), bash_repo, env + ) + py = run(py_cmd(py_repo, SCRIPT, "--json", "--dry-run", description), py_repo, env) + ps = run(ps_cmd(ps_repo, SCRIPT, "-Json", "-DryRun", description), ps_repo, env) assert bash.returncode == py.returncode == ps.returncode == 0, ( bash.stderr, py.stderr, ps.stderr, @@ -1308,4 +1675,115 @@ def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): json_stdout(py)["BRANCH_NAME"], json_stdout(ps)["BRANCH_NAME"], } - assert names == {"001-"}, names + assert names == {expected}, names + + +@pytest.mark.skipif(WINDOWS_POWERSHELL is None, reason="Windows PowerShell unavailable") +def test_windows_powershell_51_preserves_unicode_and_utf8_limit(repo: Path) -> None: + assert WINDOWS_POWERSHELL is not None + for short_name, expected in ( + ("添加用户", "001-添加用户"), + ("𠀀" * 240, "001-" + "𠀀" * 60), + ): + result = run( + [ + WINDOWS_POWERSHELL, + "-NoProfile", + "-File", + str(repo / ".specify/scripts/powershell/create-new-feature.ps1"), + "-Json", + "-DryRun", + "-ShortName", + short_name, + "x", + ], + repo, + ) + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + assert len(expected.encode("utf-8")) <= 244 + + for description, expected in ( + ("ſet account", "001-ſet-account"), + ("Set account", "001-account"), + ): + result = run( + [ + WINDOWS_POWERSHELL, + "-NoProfile", + "-File", + str(repo / ".specify/scripts/powershell/create-new-feature.ps1"), + "-Json", + "-DryRun", + description, + ], + repo, + ) + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + + +@requires_bash +@pytest.mark.parametrize("variant", ["bash", "python", "powershell"]) +@pytest.mark.parametrize( + ("short_name", "expected_suffix"), + [ + ("客" * 100, "客" * 80), + ("a" + "客" * 80, "a" + "客" * 79), + ("𠀀" * 240, "𠀀" * 60), + ], + ids=["exact_boundary", "partial_codepoint", "four_byte_long_suffix"], +) +def test_unicode_branch_name_fits_244_bytes( + repo: Path, variant: str, short_name: str, expected_suffix: str +) -> None: + """Long UTF-8 names remain valid Git refs within GitHub's byte limit.""" + if variant == "powershell" and not HAS_POWERSHELL: + pytest.skip("no PowerShell available") + powershell = variant == "powershell" + args = ( + ("-Json", "-DryRun", "-ShortName", short_name, "x") + if powershell + else ("--json", "--dry-run", "--short-name", short_name, "x") + ) + command = {"bash": bash_cmd, "python": py_cmd, "powershell": ps_cmd}[variant] + result = run(command(repo, SCRIPT, *args), repo) + + assert result.returncode == 0, result.stderr + branch = json_stdout(result)["BRANCH_NAME"] + assert branch == f"001-{expected_suffix}" + assert len(branch.encode("utf-8")) <= 244 + assert "244-byte limit" in result.stderr + assert subprocess.run( + ["git", "check-ref-format", "--branch", branch], capture_output=True + ).returncode == 0 + + +@requires_bash +def test_bash_truncation_uses_bounded_byte_checks(repo: Path, tmp_path: Path) -> None: + real_wc = shutil.which("wc") + assert real_wc is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + count_file = tmp_path / "wc-count" + shim = shim_dir / "wc" + shim.write_text( + "#!/bin/sh\n" + f"printf . >> {shlex.quote(_bash_posix_path(count_file))}\n" + f"exec {shlex.quote(_bash_posix_path(Path(real_wc)))} \"$@\"\n", + encoding="utf-8", + newline="\n", + ) + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", "𠀀" * 240, "x"), + repo, + env, + ) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-" + "𠀀" * 60 + assert count_file.read_text(encoding="utf-8").count(".") <= 16