From eb4c9dc662854cd89d713479a0d9e6d887212f92 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:49:56 +0000 Subject: [PATCH 1/9] Fix non-latin-feature-names: preserve Unicode feature names Apply the remediation from the bug assessment on issue #4574. Refs #4574 Assisted-by: GitHub Copilot (model: gpt-5.2-codex, autonomous) Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- scripts/bash/create-new-feature.sh | 13 +++---- scripts/powershell/create-new-feature.ps1 | 14 ++----- scripts/python/create_new_feature.py | 4 +- .../test_create_new_feature_python_parity.py | 37 +++++-------------- 4 files changed, 22 insertions(+), 46 deletions(-) diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index 1367014908..450f493037 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -147,16 +147,15 @@ spec_prefix_exists() { # Function to clean and format a branch name # # Three details keep this byte-identical to the Python and PowerShell twins: -# * LC_ALL=C -- in a UTF-8 locale glibc resolves the a-z *range* through -# collation, so [^a-z0-9] keeps accented lowercase letters that -# re.sub(r"[^a-z0-9]", ...) and .NET's -replace both strip. +# * LC_ALL=C -- the character classes below only replace separators, leaving +# UTF-8 letters and digits intact without depending on locale collation. # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. clean_branch_name() { local name="$1" local -x LC_ALL=C - printf '%s\n' "$name" | tr '[:upper:]' '[:lower:]' | sed 's/[^a-z0-9]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' + printf '%s\n' "$name" | tr '[:upper:]' '[:lower:]' | sed 's/[[:space:][:punct:]]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' } # Fit a feature prefix and suffix within GitHub's branch-name limit. @@ -209,12 +208,12 @@ generate_branch_name() { # Common stop words to filter out local stop_words="^(i|a|an|the|to|for|of|in|on|at|by|with|from|is|are|was|were|be|been|being|have|has|had|do|does|did|will|would|should|could|can|may|might|must|shall|this|that|these|those|my|your|our|their|want|need|add|get|set)$" - # Convert to lowercase and split into words. LC_ALL=C for the same - # collation reason documented on clean_branch_name, and so the `grep -qw` + # Convert to lowercase and split into words. LC_ALL=C keeps ASCII + # punctuation handling stable while non-ASCII words remain intact, and so the `grep -qw` # acronym probe below uses ASCII word boundaries like the Python twin's # (? Args: def _clean_branch_name(name: str) -> str: - cleaned = re.sub(r"[^a-z0-9]", "-", name.lower()) + cleaned = re.sub(r"[^\w]|_", "-", name.lower(), flags=re.UNICODE) cleaned = re.sub(r"-+", "-", cleaned) return cleaned.strip("-") def _generate_branch_name(description: str) -> str: - clean = re.sub(r"[^a-z0-9]", " ", description.lower()) + clean = re.sub(r"[^\w]|_", " ", description.lower(), flags=re.UNICODE) meaningful: list[str] = [] for word in clean.split(): if word in _STOP_WORDS: diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 36acefb7bc..6f76d40757 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -1166,15 +1166,7 @@ def test_all_variants_corrected_prefix_skips_timestamp_collision(repo: Path) -> def test_bash_branch_name_ignores_locale_collation( repo: Path, description: str ) -> None: - """Branch naming must not depend on the caller's locale. - - ``clean_branch_name``/``generate_branch_name`` sanitize with - ``sed 's/[^a-z0-9]/-/g'``. Run under a collation-ordered locale that class - keeps accented lowercase letters, so bash produced - ``001-ajouter-réservation-hôtelière`` where the Python and PowerShell twins - produce ``001-ajouter-servation-teli``: the same description yielded a - different ``specs/`` directory on two machines that differ only in ``LANG``. - """ + """Branch naming preserves Unicode letters across the script twins.""" locale_name = collation_range_locale() if locale_name is None: pytest.skip("no locale with collation-ordered [a-z] ranges available") @@ -1189,12 +1181,10 @@ def test_bash_branch_name_ignores_locale_collation( assert py.returncode == bash.returncode == 0 assert json_stdout(py) == json_stdout(bash) branch = json_stdout(bash)["BRANCH_NAME"] - assert isinstance(branch, str) and branch.isascii(), branch + assert isinstance(branch, str) and "réservation" in branch, branch # The run above reaches generate_branch_name. --short-name reaches - # clean_branch_name, a separate function carrying its own LC_ALL=C, so - # exercise the accented value through both: neither copy can then regress - # on its own without a failure here. + # clean_branch_name, a separate function, so exercise both paths. short_args = ("--json", "--dry-run", "--short-name", description, "x") bash_short = run(bash_cmd(repo, SCRIPT, *short_args), repo, env) py_short = run(py_cmd(repo, SCRIPT, *short_args), repo, env) @@ -1202,7 +1192,7 @@ def test_bash_branch_name_ignores_locale_collation( assert py_short.returncode == bash_short.returncode == 0 assert json_stdout(py_short) == json_stdout(bash_short) short_branch = json_stdout(bash_short)["BRANCH_NAME"] - assert isinstance(short_branch, str) and short_branch.isascii(), short_branch + assert isinstance(short_branch, str) and "réservation" in short_branch, short_branch @requires_bash @@ -1264,18 +1254,10 @@ def test_python_dash_prefixed_short_name_matches_bash( ["!!! ??? ***", "добавить", "添加用户"], ids=["punctuation_only", "cyrillic", "han"], ) -def test_powershell_survives_description_with_no_ascii_words( +def test_powershell_preserves_unicode_description( tmp_path: Path, description: str ): - """A description with no [a-z0-9] characters must not crash the PS twin. - - ``ConvertTo-CleanBranchName`` blanks every non-ASCII character, so the - fallback pipeline yields nothing and ``[string]::Join`` received ``$null`` - — an ArgumentNullException, made terminating by - ``$ErrorActionPreference = 'Stop'``. The script died with a .NET stack - trace and exit 1 where the bash and Python twins both return an empty - suffix. This fires for any feature phrased in a non-Latin script. - """ + """Descriptions using Unicode letters produce a usable branch suffix.""" repo = _setup_repo(tmp_path) ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) @@ -1283,13 +1265,14 @@ def test_powershell_survives_description_with_no_ascii_words( assert ps.returncode == 0, ps.stderr assert "ArgumentNullException" not in ps.stderr assert "Join" not in ps.stderr - assert json_stdout(ps)["BRANCH_NAME"] == "001-" + expected = "001-" if description.startswith("!") else f"001-{description}" + assert json_stdout(ps)["BRANCH_NAME"] == expected @requires_bash @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): - """All three twins agree on the branch name for such a description.""" + """All three twins agree on the Unicode branch name.""" description = "добавить" bash_repo = _setup_repo(tmp_path, "b") @@ -1308,4 +1291,4 @@ def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): json_stdout(py)["BRANCH_NAME"], json_stdout(ps)["BRANCH_NAME"], } - assert names == {"001-"}, names + assert names == {"001-добавить"}, names From 89cee3e338cc687e2914a63fd9f3ce22fce459b8 Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Tue, 29 Sep 2026 07:38:36 -0500 Subject: [PATCH 2/9] fix: complete Unicode feature naming across script variants Replace the initial proposed sanitizer with Unicode-aware name generation across Bash, PowerShell, and Python. Keep UTF-8 branch names within GitHub byte limits, preserve existing punctuation-only warnings, and update parity tests and documentation. Refs github/spec-kit#4574. Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/reference/core.md | 16 ++- scripts/bash/create-new-feature.sh | 60 +++++--- scripts/powershell/create-new-feature.ps1 | 48 +++++-- scripts/python/create_new_feature.py | 22 +-- .../test_create_new_feature_python_parity.py | 134 +++++++++++++----- 5 files changed, 201 insertions(+), 79 deletions(-) diff --git a/docs/reference/core.md b/docs/reference/core.md index dc9255e697..11324fa842 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -65,20 +65,22 @@ specify init my-project --integration copilot --preset compliance ## Naming Features with the Helper Scripts When calling the bundled `create-new-feature` helper scripts directly, generated -names retain only ASCII letters and digits. A description entirely in a non-Latin -script, or made only of punctuation, can therefore produce an empty suffix such -as `001-`. The scripts warn on stderr when this happens, including during a dry -run; JSON output remains parseable. +names retain Unicode letters and digits in UTF-8, so a description such as +`添加用户` produces `001-添加用户`. Descriptions made only of punctuation can still +produce an empty suffix such as `001-`; the scripts warn on stderr when this +happens, including during a dry run. JSON output remains parseable. -Keep the original description and supply a readable ASCII short name: +To choose a different name, keep the original description and supply a short name: ```bash -bash .specify/scripts/bash/create-new-feature.sh --json --short-name user-auth "添加用户" +bash .specify/scripts/bash/create-new-feature.sh --json --short-name 用户管理 "添加用户" ``` The Python helper also accepts `--short-name`; the PowerShell helper uses `-ShortName`. A supplied short name is cleaned by the same rules, so it must -contain at least one ASCII letter or digit. +contain at least one letter or digit. The Bash helper needs an installed UTF-8 +locale to recognize Unicode letters and digits. ASCII capitals are lowercased; +non-ASCII letter casing is preserved across the script variants. ## Check Installed Tools diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index 450f493037..f6cf6f3235 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -146,16 +146,35 @@ spec_prefix_exists() { # Function to clean and format a branch name # -# Three details keep this byte-identical to the Python and PowerShell twins: -# * LC_ALL=C -- the character classes below only replace separators, leaving -# UTF-8 letters and digits intact without depending on locale collation. +# Three details keep this consistent with the Python and PowerShell twins: +# * A UTF-8 locale makes [:alnum:] recognize Unicode letters and digits. # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. +UNICODE_LOCALE="" +for candidate in C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_ALL:-${LC_CTYPE:-${LANG:-}}}"; do + if [ -n "$candidate" ] && [ "$(printf 'é。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é-' ]; then + UNICODE_LOCALE="$candidate" + break + fi +done + +if [ -z "$UNICODE_LOCALE" ]; then + UNICODE_LOCALE=C + if printf '%s' "${SHORT_NAME:-$FEATURE_DESCRIPTION}" | LC_ALL=C grep -q '[^ -~]'; then + echo "Error: A UTF-8 locale is required to create a Unicode feature name" >&2 + exit 1 + fi +fi + clean_branch_name() { local name="$1" - local -x LC_ALL=C - printf '%s\n' "$name" | tr '[:upper:]' '[:lower:]' | sed 's/[[:space:][:punct:]]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' + local -x LC_ALL="$UNICODE_LOCALE" + printf '%s\n' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | sed 's/[^[:alnum:]]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' +} + +branch_byte_count() { + printf '%s' "$1" | wc -c | tr -d '[:space:]' } # Fit a feature prefix and suffix within GitHub's branch-name limit. @@ -164,11 +183,16 @@ fit_branch_name() { local branch_suffix="$2" local branch_name="${feature_num}-${branch_suffix}" - if [ ${#branch_name} -gt $MAX_BRANCH_LENGTH ]; then + if [ "$(branch_byte_count "$branch_name")" -gt "$MAX_BRANCH_LENGTH" ]; then local prefix_length=$(( ${#feature_num} + 1 )) local max_suffix_length=$((MAX_BRANCH_LENGTH - prefix_length)) local truncated_suffix - truncated_suffix=$(printf '%s' "$branch_suffix" | cut -c "1-$max_suffix_length" | sed 's/-$//') + local -x LC_ALL="$UNICODE_LOCALE" + truncated_suffix="${branch_suffix:0:$max_suffix_length}" + while [ "$(branch_byte_count "$truncated_suffix")" -gt "$max_suffix_length" ]; do + truncated_suffix="${truncated_suffix%?}" + done + truncated_suffix="${truncated_suffix%-}" branch_name="${feature_num}-${truncated_suffix}" fi @@ -208,12 +232,10 @@ generate_branch_name() { # Common stop words to filter out local stop_words="^(i|a|an|the|to|for|of|in|on|at|by|with|from|is|are|was|were|be|been|being|have|has|had|do|does|did|will|would|should|could|can|may|might|must|shall|this|that|these|those|my|your|our|their|want|need|add|get|set)$" - # Convert to lowercase and split into words. LC_ALL=C keeps ASCII - # punctuation handling stable while non-ASCII words remain intact, and so the `grep -qw` - # acronym probe below uses ASCII word boundaries like the Python twin's - # (?= 3 OR are potential acronyms) + # Retain non-ASCII words even when shorter than three characters. if ! echo "$word" | grep -qiE "$stop_words"; then - if [ ${#word} -ge 3 ]; then + if [ ${#word} -ge 3 ] || printf '%s' "$word" | LC_ALL=C grep -q '[^ -~]'; then meaningful_words+=("$word") # Keep short words that appear as an uppercase acronym in the original. # Uppercase via tr and match with grep -w (both portable) rather than # bash's 4+ "^^" case expansion (breaks on macOS bash 3.2) and \b (non-POSIX). - elif printf '%s' "$description" | grep -qw -- "$(printf '%s' "$word" | tr '[:lower:]' '[:upper:]')"; then + elif printf '%s' "$description" | LC_ALL=C grep -qw -- "$(printf '%s' "$word" | LC_ALL=C tr '[:lower:]' '[:upper:]')"; then meaningful_words+=("$word") fi fi @@ -265,7 +287,7 @@ else fi if [ -z "$BRANCH_SUFFIX" ]; then - echo "[specify] Warning: Feature name is empty after removing unsupported characters. Use --short-name with ASCII letters or digits (for example, user-auth)." >&2 + echo "[specify] Warning: Feature name is empty after removing unsupported characters. Use --short-name with letters or digits (for example, user-auth)." >&2 fi # Warn if --number and --timestamp are both specified @@ -338,8 +360,8 @@ ORIGINAL_BRANCH_NAME="${FEATURE_NUM}-${BRANCH_SUFFIX}" BRANCH_NAME=$(fit_branch_name "$FEATURE_NUM" "$BRANCH_SUFFIX") if [ "$BRANCH_NAME" != "$ORIGINAL_BRANCH_NAME" ]; then >&2 echo "[specify] Warning: Branch name exceeded GitHub's 244-byte limit" - >&2 echo "[specify] Original: $ORIGINAL_BRANCH_NAME (${#ORIGINAL_BRANCH_NAME} bytes)" - >&2 echo "[specify] Truncated to: $BRANCH_NAME (${#BRANCH_NAME} bytes)" + >&2 echo "[specify] Original: $ORIGINAL_BRANCH_NAME ($(branch_byte_count "$ORIGINAL_BRANCH_NAME") bytes)" + >&2 echo "[specify] Truncated to: $BRANCH_NAME ($(branch_byte_count "$BRANCH_NAME") bytes)" fi FEATURE_DIR="$SPECS_DIR/$BRANCH_NAME" diff --git a/scripts/powershell/create-new-feature.ps1 b/scripts/powershell/create-new-feature.ps1 index f6c6b92880..aa4c0e5801 100644 --- a/scripts/powershell/create-new-feature.ps1 +++ b/scripts/powershell/create-new-feature.ps1 @@ -83,10 +83,32 @@ function Test-SpecPrefixInUse { Select-Object -First 1) } +function ConvertTo-AsciiLower { + param([string]$Name) + + return [regex]::Replace($Name, '[A-Z]', { param($match) $match.Value.ToLowerInvariant() }) +} + +function ConvertTo-UnicodeWords { + param([string]$Name, [string]$Separator) + + $lowerName = ConvertTo-AsciiLower -Name $Name + return [regex]::Replace($lowerName, '[\uD800-\uDBFF][\uDC00-\uDFFF]|[^\p{L}\p{N}]', { + param($match) + if ($match.Length -eq 2) { + $category = [System.Globalization.CharUnicodeInfo]::GetUnicodeCategory($match.Value, 0) + if ($category.ToString() -match '(Letter|Number)$') { + return $match.Value + } + } + return $Separator + }) +} + function ConvertTo-CleanBranchName { param([string]$Name) - return $Name.ToLower() -replace '[^\p{L}\p{N}]', '-' -replace '-{2,}', '-' -replace '^-', '' -replace '-$', '' + return (ConvertTo-UnicodeWords -Name $Name -Separator '-') -replace '-{2,}', '-' -replace '^-', '' -replace '-$', '' } function Get-FittedBranchName { @@ -96,10 +118,15 @@ function Get-FittedBranchName { ) $fittedName = "$FeatureNum-$BranchSuffix" - if ($fittedName.Length -gt $maxBranchLength) { + if ([System.Text.Encoding]::UTF8.GetByteCount($fittedName) -gt $maxBranchLength) { $prefixLength = $FeatureNum.Length + 1 $maxSuffixLength = $maxBranchLength - $prefixLength - $truncatedSuffix = $BranchSuffix.Substring(0, [Math]::Min($BranchSuffix.Length, $maxSuffixLength)) + $bytes = [System.Text.Encoding]::UTF8.GetBytes($BranchSuffix) + $bytesToUse = $maxSuffixLength + while ($bytesToUse -gt 0 -and ($bytes[$bytesToUse] -band 0xC0) -eq 0x80) { + $bytesToUse-- + } + $truncatedSuffix = [System.Text.Encoding]::UTF8.GetString($bytes, 0, $bytesToUse) $truncatedSuffix = $truncatedSuffix -replace '-$', '' $fittedName = "$FeatureNum-$truncatedSuffix" } @@ -132,8 +159,8 @@ function Get-BranchName { 'want', 'need', 'add', 'get', 'set' ) - # Convert to lowercase and extract Unicode words. - $cleanName = $Description.ToLower() -replace '[^\p{L}\p{N}\s]', ' ' + # Lowercase ASCII and extract Unicode words, matching the shell variant. + $cleanName = ConvertTo-UnicodeWords -Name $Description -Separator ' ' $words = $cleanName -split '\s+' | Where-Object { $_ } # Filter words: remove stop words and words shorter than 3 chars (unless they're uppercase acronyms in original) @@ -142,8 +169,9 @@ function Get-BranchName { # Skip stop words if ($stopWords -contains $word) { continue } - # Keep words that are length >= 3 OR appear as uppercase in original (likely acronyms) - if ($word.Length -ge 3) { + # Keep Unicode words even when short; ASCII words still need three + # characters or an uppercase acronym in the original. + if ($word.Length -ge 3 -or $word -match '[^\x00-\x7F]') { $meaningfulWords += $word } elseif ($Description -cmatch "(? str: _MAX_BRANCH_LENGTH = 244 _MAX_FEATURE_NUMBER = 2**63 - 1 +_ASCII_LOWER = str.maketrans( + "ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz" +) def _int64_from_digits(value: str) -> int | None: @@ -167,18 +170,18 @@ def _parse_args(argv: list[str], argv0: str) -> Args: def _clean_branch_name(name: str) -> str: - cleaned = re.sub(r"[^\w]|_", "-", name.lower(), flags=re.UNICODE) + cleaned = re.sub(r"[^\w]|_", "-", name.translate(_ASCII_LOWER)) cleaned = re.sub(r"-+", "-", cleaned) return cleaned.strip("-") def _generate_branch_name(description: str) -> str: - clean = re.sub(r"[^\w]|_", " ", description.lower(), flags=re.UNICODE) + clean = re.sub(r"[^\w]|_", " ", description.translate(_ASCII_LOWER)) meaningful: list[str] = [] for word in clean.split(): if word in _STOP_WORDS: continue - if len(word) >= 3: + if len(word) >= 3 or not word.isascii(): meaningful.append(word) # Keep short words that appear as an uppercase acronym in the original, # mirroring the bash twin's case-sensitive `grep -qw` check. @@ -217,11 +220,13 @@ def _get_highest_from_specs(specs_dir: Path) -> int: def _fit_branch_name(feature_num: str, branch_suffix: str) -> str: """Fit a feature prefix and suffix within GitHub's branch-name limit.""" branch_name = f"{feature_num}-{branch_suffix}" - if len(branch_name) <= _MAX_BRANCH_LENGTH: + if len(branch_name.encode("utf-8")) <= _MAX_BRANCH_LENGTH: return branch_name max_suffix_length = _MAX_BRANCH_LENGTH - (len(feature_num) + 1) - truncated_suffix = re.sub(r"-$", "", branch_suffix[:max_suffix_length]) + truncated_suffix = branch_suffix.encode("utf-8")[:max_suffix_length].decode( + "utf-8", errors="ignore" + ).rstrip("-") return f"{feature_num}-{truncated_suffix}" @@ -268,7 +273,7 @@ def main(argv: list[str] | None = None) -> int: if not branch_suffix: print( "[specify] Warning: Feature name is empty after removing unsupported characters. " - "Use --short-name with ASCII letters or digits (for example, user-auth).", + "Use --short-name with letters or digits (for example, user-auth).", file=sys.stderr, ) @@ -363,11 +368,12 @@ def main(argv: list[str] | None = None) -> int: ) print( f"[specify] Original: {original_branch_name} " - f"({len(original_branch_name)} bytes)", + f"({len(original_branch_name.encode('utf-8'))} bytes)", file=sys.stderr, ) print( - f"[specify] Truncated to: {branch_name} ({len(branch_name)} bytes)", + f"[specify] Truncated to: {branch_name} " + f"({len(branch_name.encode('utf-8'))} bytes)", file=sys.stderr, ) diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 6f76d40757..21dbfde1b0 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -3,6 +3,7 @@ from __future__ import annotations import re +import subprocess from pathlib import Path import pytest @@ -72,11 +73,11 @@ def repo_pair(tmp_path: Path) -> tuple[Path, Path]: @pytest.mark.parametrize( "description,short_name,suffix,warns", [ - ("添加用户", None, "", True), - ("добавить", None, "", True), + ("添加用户", None, "添加用户", False), + ("добавить", None, "добавить", False), ("!!! ??? ***", None, "", True), ("添加用户", "user-auth", "user-auth", False), - ("Add users", "用户", "", True), + ("Add users", "用户", "用户", False), ("Add user authentication", None, "user-authentication", False), ], ) @@ -89,7 +90,7 @@ def test_empty_feature_name_warning( suffix: str, warns: bool, ) -> None: - """Report unusable names without changing JSON or feature creation (#4574).""" + """Preserve UTF-8 names and warn only when no name remains (#4574).""" powershell = variant == "powershell" args = ["-Json" if powershell else "--json"] if dry_run: @@ -107,7 +108,7 @@ def test_empty_feature_name_warning( assert result.stderr.count(warning) == int(warns) if warns: assert ("-ShortName" if powershell else "--short-name") in result.stderr - assert "ASCII letters or digits" in result.stderr + assert "letters or digits" in result.stderr assert (repo / "specs" / f"001-{suffix}" / "spec.md").exists() is not dry_run @@ -173,10 +174,8 @@ def deny_listing(_path: Path): "I want to add the new API rate limiting feature for users", "Fix UI for DB sync", "a to the of", - # An acronym touching an accented letter: bash probes with `grep -qw` - # under LC_ALL=C, where the accent is a word boundary, so the Python - # twin must use explicit ASCII lookarounds rather than a Unicode \b. "Fix \u00e9DB\u00e9 sync", + "Ajouter la réservation hôtelière", ], ids=[ "plain", @@ -184,6 +183,7 @@ def deny_listing(_path: Path): "acronyms", "all_stop_words_fallback", "acronym_next_to_non_ascii", + "accented_words", ], ) def test_python_branch_name_generation_matches_bash( @@ -199,22 +199,25 @@ def test_python_branch_name_generation_matches_bash( @requires_bash @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") -def test_all_variants_keep_acronym_next_to_non_ascii(repo: Path) -> None: - """An acronym touching an accented letter survives in all three twins. - - bash probes for acronyms with `grep -qw` under LC_ALL=C, where an accented - letter is a non-word byte and therefore a boundary. Python's \\b and .NET's - \\b are Unicode-aware and saw "\u00e9DB\u00e9" as a single word, dropping the - acronym; all three now spell the boundary out as ASCII. - """ - description = "Fix \u00e9DB\u00e9 sync" +@pytest.mark.parametrize( + ("description", "expected"), + [ + ("Fix éDBé sync", "001-fix-édbé-sync"), + ("É DB sync", "001-É-db-sync"), + ("é DB sync", "001-é-db-sync"), + ], +) +def test_all_variants_keep_acronym_next_to_non_ascii( + repo: Path, description: str, expected: str +) -> None: + """Unicode words and adjacent ASCII acronyms survive together.""" bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) assert bash.returncode == py.returncode == ps.returncode == 0 assert json_stdout(py) == json_stdout(bash) == json_stdout(ps) - assert json_stdout(ps)["BRANCH_NAME"] == "001-fix-db-sync" + assert json_stdout(ps)["BRANCH_NAME"] == expected @requires_bash @@ -1166,7 +1169,7 @@ def test_all_variants_corrected_prefix_skips_timestamp_collision(repo: Path) -> def test_bash_branch_name_ignores_locale_collation( repo: Path, description: str ) -> None: - """Branch naming preserves Unicode letters across the script twins.""" + """Branch naming preserves each description's Unicode words across locales.""" locale_name = collation_range_locale() if locale_name is None: pytest.skip("no locale with collation-ordered [a-z] ranges available") @@ -1181,10 +1184,13 @@ def test_bash_branch_name_ignores_locale_collation( assert py.returncode == bash.returncode == 0 assert json_stdout(py) == json_stdout(bash) branch = json_stdout(bash)["BRANCH_NAME"] - assert isinstance(branch, str) and "réservation" in branch, branch + assert isinstance(branch, str) and branch == { + "Añadir autenticación de usuario": "001-añadir-autenticación-usuario", + "Prüfung für Benutzer anlegen": "001-prüfung-für-benutzer-anlegen", + "Ajouter la réservation hôtelière": "001-ajouter-réservation-hôtelière", + }[description] - # The run above reaches generate_branch_name. --short-name reaches - # clean_branch_name, a separate function, so exercise both paths. + # Both the generated and explicit-name paths must retain the input's words. short_args = ("--json", "--dry-run", "--short-name", description, "x") bash_short = run(bash_cmd(repo, SCRIPT, *short_args), repo, env) py_short = run(py_cmd(repo, SCRIPT, *short_args), repo, env) @@ -1192,7 +1198,11 @@ def test_bash_branch_name_ignores_locale_collation( assert py_short.returncode == bash_short.returncode == 0 assert json_stdout(py_short) == json_stdout(bash_short) short_branch = json_stdout(bash_short)["BRANCH_NAME"] - assert isinstance(short_branch, str) and "réservation" in short_branch, short_branch + assert isinstance(short_branch, str) and short_branch == { + "Añadir autenticación de usuario": "001-añadir-autenticación-de-usuario", + "Prüfung für Benutzer anlegen": "001-prüfung-für-benutzer-anlegen", + "Ajouter la réservation hôtelière": "001-ajouter-la-réservation-hôtelière", + }[description] @requires_bash @@ -1250,14 +1260,18 @@ def test_python_dash_prefixed_short_name_matches_bash( @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") @pytest.mark.parametrize( - "description", - ["!!! ??? ***", "добавить", "添加用户"], + ("description", "expected"), + [ + ("!!! ??? ***", "001-"), + ("добавить", "001-добавить"), + ("添加用户", "001-添加用户"), + ], ids=["punctuation_only", "cyrillic", "han"], ) def test_powershell_preserves_unicode_description( - tmp_path: Path, description: str + tmp_path: Path, description: str, expected: str ): - """Descriptions using Unicode letters produce a usable branch suffix.""" + """Non-Latin descriptions are usable; punctuation alone still warns.""" repo = _setup_repo(tmp_path) ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) @@ -1265,23 +1279,38 @@ def test_powershell_preserves_unicode_description( assert ps.returncode == 0, ps.stderr assert "ArgumentNullException" not in ps.stderr assert "Join" not in ps.stderr - expected = "001-" if description.startswith("!") else f"001-{description}" assert json_stdout(ps)["BRANCH_NAME"] == expected @requires_bash @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") -def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): - """All three twins agree on the Unicode branch name.""" - description = "добавить" +@pytest.mark.parametrize( + ("description", "expected"), + [ + ("добавить", "001-добавить"), + ("给倒推引擎加正推能力", "001-给倒推引擎加正推能力"), + ("客户邮件。审核队列", "001-客户邮件-审核队列"), + ("ПРИВЕТ", "001-ПРИВЕТ"), + ("𠀀𠀁。功能", "001-𠀀𠀁-功能"), + ("😀!!!", "001-"), + ], +) +def test_no_ascii_word_description_matches_across_twins( + tmp_path: Path, description: str, expected: str +): + """All three twins agree on UTF-8 feature names and separators.""" bash_repo = _setup_repo(tmp_path, "b") py_repo = _setup_repo(tmp_path, "p") ps_repo = _setup_repo(tmp_path, "s") - bash = run(bash_cmd(bash_repo, SCRIPT, "--json", "--dry-run", description), bash_repo) - py = run(py_cmd(py_repo, SCRIPT, "--json", "--dry-run", description), py_repo) - ps = run(ps_cmd(ps_repo, SCRIPT, "-Json", "-DryRun", description), ps_repo) + env = clean_env() + env["LC_ALL"] = env["LANG"] = "C" + bash = run( + bash_cmd(bash_repo, SCRIPT, "--json", "--dry-run", description), bash_repo, env + ) + py = run(py_cmd(py_repo, SCRIPT, "--json", "--dry-run", description), py_repo, env) + ps = run(ps_cmd(ps_repo, SCRIPT, "-Json", "-DryRun", description), ps_repo, env) assert bash.returncode == py.returncode == ps.returncode == 0, ( bash.stderr, py.stderr, ps.stderr, @@ -1291,4 +1320,39 @@ def test_no_ascii_word_description_matches_across_twins(tmp_path: Path): json_stdout(py)["BRANCH_NAME"], json_stdout(ps)["BRANCH_NAME"], } - assert names == {"001-добавить"}, names + assert names == {expected}, names + + +@requires_bash +@pytest.mark.parametrize("variant", ["bash", "python", "powershell"]) +@pytest.mark.parametrize( + ("short_name", "expected_suffix"), + [ + ("客" * 100, "客" * 80), + ("a" + "客" * 80, "a" + "客" * 79), + ], + ids=["exact_boundary", "partial_codepoint"], +) +def test_unicode_branch_name_fits_244_bytes( + repo: Path, variant: str, short_name: str, expected_suffix: str +) -> None: + """Long UTF-8 names remain valid Git refs within GitHub's byte limit.""" + if variant == "powershell" and not HAS_POWERSHELL: + pytest.skip("no PowerShell available") + powershell = variant == "powershell" + args = ( + ("-Json", "-DryRun", "-ShortName", short_name, "x") + if powershell + else ("--json", "--dry-run", "--short-name", short_name, "x") + ) + command = {"bash": bash_cmd, "python": py_cmd, "powershell": ps_cmd}[variant] + result = run(command(repo, SCRIPT, *args), repo) + + assert result.returncode == 0, result.stderr + branch = json_stdout(result)["BRANCH_NAME"] + assert branch == f"001-{expected_suffix}" + assert len(branch.encode("utf-8")) <= 244 + assert "244-byte limit" in result.stderr + assert subprocess.run( + ["git", "check-ref-format", "--branch", branch], capture_output=True + ).returncode == 0 From 461d1583e83acb38b98294b00b844381c22e2f80 Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:27:31 -0500 Subject: [PATCH 3/9] fix: align Unicode categories and UTF-8 feature output Use Unicode letters and decimal digits consistently for feature names. Emit Python output as UTF-8 on Windows, decode parity subprocesses as UTF-8, and test the no-locale error and non-decimal number cases. Refs github/spec-kit#4574. Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/reference/core.md | 2 +- scripts/bash/create-new-feature.sh | 4 +- scripts/powershell/create-new-feature.ps1 | 4 +- scripts/python/create_new_feature.py | 13 ++- tests/parity_helpers.py | 2 + .../test_create_new_feature_python_parity.py | 101 ++++++++++++++++++ 6 files changed, 119 insertions(+), 7 deletions(-) diff --git a/docs/reference/core.md b/docs/reference/core.md index 11324fa842..b2abcf9629 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -65,7 +65,7 @@ specify init my-project --integration copilot --preset compliance ## Naming Features with the Helper Scripts When calling the bundled `create-new-feature` helper scripts directly, generated -names retain Unicode letters and digits in UTF-8, so a description such as +names retain Unicode letters and decimal digits in UTF-8, so a description such as `添加用户` produces `001-添加用户`. Descriptions made only of punctuation can still produce an empty suffix such as `001-`; the scripts warn on stderr when this happens, including during a dry run. JSON output remains parseable. diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index f6cf6f3235..db6c5727cd 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -147,13 +147,13 @@ spec_prefix_exists() { # Function to clean and format a branch name # # Three details keep this consistent with the Python and PowerShell twins: -# * A UTF-8 locale makes [:alnum:] recognize Unicode letters and digits. +# * A UTF-8 locale makes [:alnum:] recognize Unicode letters and decimal digits. # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. UNICODE_LOCALE="" for candidate in C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_ALL:-${LC_CTYPE:-${LANG:-}}}"; do - if [ -n "$candidate" ] && [ "$(printf 'é。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é-' ]; then + if [ -n "$candidate" ] && [ "$(printf 'é٥²Ⅻ。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é٥---' ]; then UNICODE_LOCALE="$candidate" break fi diff --git a/scripts/powershell/create-new-feature.ps1 b/scripts/powershell/create-new-feature.ps1 index aa4c0e5801..8688332375 100644 --- a/scripts/powershell/create-new-feature.ps1 +++ b/scripts/powershell/create-new-feature.ps1 @@ -93,11 +93,11 @@ function ConvertTo-UnicodeWords { param([string]$Name, [string]$Separator) $lowerName = ConvertTo-AsciiLower -Name $Name - return [regex]::Replace($lowerName, '[\uD800-\uDBFF][\uDC00-\uDFFF]|[^\p{L}\p{N}]', { + return [regex]::Replace($lowerName, '[\uD800-\uDBFF][\uDC00-\uDFFF]|[^\p{L}\p{Nd}]', { param($match) if ($match.Length -eq 2) { $category = [System.Globalization.CharUnicodeInfo]::GetUnicodeCategory($match.Value, 0) - if ($category.ToString() -match '(Letter|Number)$') { + if ($category.ToString() -match '(Letter|DecimalDigitNumber)$') { return $match.Value } } diff --git a/scripts/python/create_new_feature.py b/scripts/python/create_new_feature.py index 9b92c6bf16..2099a73708 100644 --- a/scripts/python/create_new_feature.py +++ b/scripts/python/create_new_feature.py @@ -170,13 +170,20 @@ def _parse_args(argv: list[str], argv0: str) -> Args: def _clean_branch_name(name: str) -> str: - cleaned = re.sub(r"[^\w]|_", "-", name.translate(_ASCII_LOWER)) + cleaned = _unicode_words(name, "-") cleaned = re.sub(r"-+", "-", cleaned) return cleaned.strip("-") +def _unicode_words(name: str, separator: str) -> str: + return "".join( + char if char.isalpha() or char.isdecimal() else separator + for char in name.translate(_ASCII_LOWER) + ) + + def _generate_branch_name(description: str) -> str: - clean = re.sub(r"[^\w]|_", " ", description.translate(_ASCII_LOWER)) + clean = _unicode_words(description, " ") meaningful: list[str] = [] for word in clean.split(): if word in _STOP_WORDS: @@ -454,4 +461,6 @@ def main(argv: list[str] | None = None) -> int: if __name__ == "__main__": + sys.stdout.reconfigure(encoding="utf-8") + sys.stderr.reconfigure(encoding="utf-8") raise SystemExit(main()) diff --git a/tests/parity_helpers.py b/tests/parity_helpers.py index bc6edc0aca..56bee0806b 100644 --- a/tests/parity_helpers.py +++ b/tests/parity_helpers.py @@ -208,6 +208,7 @@ def collation_range_locale() -> str | None: input="é\n", capture_output=True, text=True, + encoding="utf-8", check=False, env=env, ) @@ -235,6 +236,7 @@ def run( cwd=repo, capture_output=True, text=True, + encoding="utf-8", check=False, env=env if env is not None else clean_env(), timeout=timeout, diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 21dbfde1b0..6c59419e1e 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -3,6 +3,8 @@ from __future__ import annotations import re +import shlex +import shutil import subprocess from pathlib import Path @@ -112,6 +114,105 @@ def test_empty_feature_name_warning( assert (repo / "specs" / f"001-{suffix}" / "spec.md").exists() is not dry_run +@pytest.mark.parametrize("json_mode", [False, True], ids=["text", "json"]) +@pytest.mark.parametrize("dry_run", [False, True], ids=["create", "dry_run"]) +def test_python_outputs_unicode_when_default_encoding_is_cp1252( + repo: Path, json_mode: bool, dry_run: bool +) -> None: + env = clean_env() + env["PYTHONIOENCODING"] = "cp1252" + args = [] + if json_mode: + args.append("--json") + if dry_run: + args.append("--dry-run") + args.append("添加用户") + + result = run(py_cmd(repo, SCRIPT, *args), repo, env) + + assert result.returncode == 0, result.stderr + assert "001-添加用户" in result.stdout + if not dry_run: + assert "001-添加用户" in result.stderr + if json_mode: + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + else: + assert "BRANCH_NAME: 001-添加用户" in result.stdout + + +@pytest.mark.parametrize( + "variant", + [ + pytest.param("bash", marks=requires_bash), + "python", + pytest.param( + "powershell", + marks=pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available"), + ), + ], +) +@pytest.mark.parametrize( + ("short_name", "expected"), + [ + ("x²", "001-x"), + ("xⅫ", "001-x"), + ("x٥", "001-x٥"), + ("x𝟘", "001-x𝟘"), + ], + ids=["superscript_number", "roman_numeral", "decimal_digit", "supplementary_decimal"], +) +def test_unicode_number_categories_match( + repo: Path, variant: str, short_name: str, expected: str +) -> None: + powershell = variant == "powershell" + args = ( + ("-Json", "-DryRun", "-ShortName", short_name, "x") + if powershell + else ("--json", "--dry-run", "--short-name", short_name, "x") + ) + command = {"bash": bash_cmd, "python": py_cmd, "powershell": ps_cmd}[variant] + + result = run(command(repo, SCRIPT, *args), repo) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + + +@requires_bash +def test_bash_reports_missing_utf8_locale_for_unicode_only( + repo: Path, tmp_path: Path +) -> None: + real_sed = shutil.which("sed") + assert real_sed is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + sed_shim = shim_dir / "sed" + sed_shim.write_text( + "#!/bin/sh\n" + """if [ "$1" = 's/[^[:alnum:]]/-/g' ]; then exit 1; fi\n""" + f"exec {shlex.quote(real_sed)} \"$@\"\n", + encoding="utf-8", + ) + sed_shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}:{env['PATH']}" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + for args in (("添加用户",), ("--short-name", "用户", "Add users")): + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", *args), repo, env) + assert result.returncode == 1 + assert result.stdout == "" + assert "Error: A UTF-8 locale is required" in result.stderr + assert not (repo / "specs").exists() + + def _run_all_variants_allow_existing( repo: Path, *, number: str, short_name: str ): From f2c0f9380a016e24615676b28f41558a698110c5 Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:41:30 -0500 Subject: [PATCH 4/9] fix: classify Unicode consistently on Bash across locales Use Python 3 for non-ASCII character classification where POSIX locale classes vary by platform. Preserve the shell-only ASCII path, report a clear error when a Unicode name lacks Python, and cover both paths. Refs github/spec-kit#4574. Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/reference/core.md | 7 ++-- scripts/bash/create-new-feature.sh | 41 +++++++++++++++---- .../test_create_new_feature_python_parity.py | 29 +++++++++++++ 3 files changed, 67 insertions(+), 10 deletions(-) diff --git a/docs/reference/core.md b/docs/reference/core.md index b2abcf9629..ba223b3c56 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -78,9 +78,10 @@ bash .specify/scripts/bash/create-new-feature.sh --json --short-name 用户管 The Python helper also accepts `--short-name`; the PowerShell helper uses `-ShortName`. A supplied short name is cleaned by the same rules, so it must -contain at least one letter or digit. The Bash helper needs an installed UTF-8 -locale to recognize Unicode letters and digits. ASCII capitals are lowercased; -non-ASCII letter casing is preserved across the script variants. +contain at least one letter or digit. For non-ASCII names, the Bash helper needs +an installed UTF-8 locale and a Python 3 interpreter for Unicode classification. +ASCII capitals are lowercased; non-ASCII letter casing is preserved across the +script variants. ## Check Installed Tools diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index db6c5727cd..0c37668b43 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -147,13 +147,13 @@ spec_prefix_exists() { # Function to clean and format a branch name # # Three details keep this consistent with the Python and PowerShell twins: -# * A UTF-8 locale makes [:alnum:] recognize Unicode letters and decimal digits. +# * Unicode classification uses Python: POSIX [:alnum:] differs by platform. # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. UNICODE_LOCALE="" for candidate in C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_ALL:-${LC_CTYPE:-${LANG:-}}}"; do - if [ -n "$candidate" ] && [ "$(printf 'é٥²Ⅻ。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é٥---' ]; then + if [ -n "$candidate" ] && [ "$(printf 'é。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é-' ]; then UNICODE_LOCALE="$candidate" break fi @@ -167,10 +167,37 @@ if [ -z "$UNICODE_LOCALE" ]; then fi fi +unicode_words() { + local name="$1" + local separator="$2" + if printf '%s' "$name" | LC_ALL=C grep -q '[^ -~]'; then + local python_spec + if ! python_spec=$(_python3_command); then + echo "Error: Python 3 is required to create a Unicode feature name" >&2 + return 1 + fi + local -a python_cmd + read -r -a python_cmd <<< "$python_spec" + printf '%s' "$name" | "${python_cmd[@]}" -c ' +import sys +value = sys.stdin.buffer.read().decode("utf-8") +lower = str.maketrans("ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz") +result = "".join( + char if char.isalpha() or char.isdecimal() else sys.argv[1] + for char in value.translate(lower) +) +sys.stdout.buffer.write(result.encode("utf-8")) +' "$separator" + else + printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed "s/[^a-z0-9]/$separator/g" + fi +} + clean_branch_name() { local name="$1" - local -x LC_ALL="$UNICODE_LOCALE" - printf '%s\n' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | sed 's/[^[:alnum:]]/-/g' | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' + local cleaned + cleaned=$(unicode_words "$name" '-') || return 1 + printf '%s\n' "$cleaned" | sed 's/--*/-/g' | sed 's/^-//' | sed 's/-$//' } branch_byte_count() { @@ -232,10 +259,10 @@ generate_branch_name() { # Common stop words to filter out local stop_words="^(i|a|an|the|to|for|of|in|on|at|by|with|from|is|are|was|were|be|been|being|have|has|had|do|does|did|will|would|should|could|can|may|might|must|shall|this|that|these|those|my|your|our|their|want|need|add|get|set)$" - # Use a UTF-8 locale for Unicode character classes. Lowercase only ASCII - # so results do not depend on the platform's multibyte case conversion. + # Use a UTF-8 locale for character-safe length checks and split words. local -x LC_ALL="$UNICODE_LOCALE" - local clean_name=$(printf '%s\n' "$description" | LC_ALL=C tr '[:upper:]' '[:lower:]' | sed 's/[^[:alnum:]]/ /g') + local clean_name + clean_name=$(unicode_words "$description" ' ') || return 1 # Filter words: remove stop words and words shorter than 3 chars (unless they're uppercase acronyms in original) local meaningful_words=() diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 6c59419e1e..835bf6f889 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -213,6 +213,35 @@ def test_bash_reports_missing_utf8_locale_for_unicode_only( assert not (repo / "specs").exists() +@requires_bash +def test_bash_requires_python_only_for_unicode_names( + repo: Path, tmp_path: Path +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}:{env['PATH']}" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + unicode_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env + ) + assert unicode_result.returncode == 1 + assert unicode_result.stdout == "" + assert "Error: Python 3 is required to create a Unicode feature name" in unicode_result.stderr + + def _run_all_variants_allow_existing( repo: Path, *, number: str, short_name: str ): From f23ccf49138e26295a1007ea2ecc158094e066c4 Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Wed, 30 Sep 2026 07:54:12 -0500 Subject: [PATCH 5/9] fix: select Python reliably for Unicode Bash feature names Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- scripts/bash/create-new-feature.sh | 16 ++++-- .../test_create_new_feature_python_parity.py | 54 +++++++++++++++++++ 2 files changed, 66 insertions(+), 4 deletions(-) diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index 0c37668b43..ad944a475b 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -171,13 +171,21 @@ unicode_words() { local name="$1" local separator="$2" if printf '%s' "$name" | LC_ALL=C grep -q '[^ -~]'; then - local python_spec - if ! python_spec=$(_python3_command); then + local -a python_cmd=() + local override="${SPECKIT_PYTHON_EXECUTABLE:-${SPECKIT_PYTHON:-}}" + if [ -n "$override" ] && command -v "$override" >/dev/null 2>&1 && + "$override" -c 'import sys; raise SystemExit(sys.version_info.major != 3)' >/dev/null 2>&1; then + python_cmd=("$override") + else + local python_line + while IFS= read -r python_line; do + python_cmd+=("$python_line") + done < <(_python3_command) + fi + if [ "${#python_cmd[@]}" -eq 0 ]; then echo "Error: Python 3 is required to create a Unicode feature name" >&2 return 1 fi - local -a python_cmd - read -r -a python_cmd <<< "$python_spec" printf '%s' "$name" | "${python_cmd[@]}" -c ' import sys value = sys.stdin.buffer.read().decode("utf-8") diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 835bf6f889..e60d850756 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -2,10 +2,12 @@ from __future__ import annotations +import os import re import shlex import shutil import subprocess +import sys from pathlib import Path import pytest @@ -15,6 +17,7 @@ from tests.conftest import requires_bash from tests.parity_helpers import ( HAS_POWERSHELL, + _bash_posix_path, bash_cmd, break_wrap_layer, clean_env, @@ -23,6 +26,7 @@ install_scripts, json_stdout, make_repo, + make_yaml_less_venv, normalize_repo_paths, normalize_script_names, ps_cmd, @@ -242,6 +246,56 @@ def test_bash_requires_python_only_for_unicode_names( assert "Error: Python 3 is required to create a Unicode feature name" in unicode_result.stderr +@requires_bash +def test_bash_unicode_uses_configured_python_without_pyyaml_in_spaced_path( + repo: Path, tmp_path: Path +) -> None: + no_yaml_exe = make_yaml_less_venv(tmp_path / "tool env") + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["SPECKIT_PYTHON_EXECUTABLE"] = _bash_posix_path(no_yaml_exe) + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + + +@requires_bash +def test_bash_unicode_uses_py_launcher_with_separate_version_arg( + repo: Path, tmp_path: Path +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + launcher = shim_dir / "py" + launcher.write_text( + "#!/bin/sh\n" + '[ "$1" = "-3" ] || exit 1\n' + "shift\n" + f'exec {shlex.quote(_bash_posix_path(Path(sys.executable)))} "$@"\n', + encoding="utf-8", + newline="\n", + ) + launcher.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "添加用户"), repo, env) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-添加用户" + + def _run_all_variants_allow_existing( repo: Path, *, number: str, short_name: str ): From 68d800b6eb185d85acd0c7668cd000eeec4e1a4b Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Wed, 30 Sep 2026 09:33:51 -0500 Subject: [PATCH 6/9] fix: respect LC_ALL and preserve Unicode stop words in Bash Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/reference/core.md | 4 + scripts/bash/create-new-feature.sh | 14 +++- .../test_create_new_feature_python_parity.py | 77 ++++++++++++++++++- 3 files changed, 91 insertions(+), 4 deletions(-) diff --git a/docs/reference/core.md b/docs/reference/core.md index ba223b3c56..85ce7881ae 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -80,6 +80,10 @@ The Python helper also accepts `--short-name`; the PowerShell helper uses `-ShortName`. A supplied short name is cleaned by the same rules, so it must contain at least one letter or digit. For non-ASCII names, the Bash helper needs an installed UTF-8 locale and a Python 3 interpreter for Unicode classification. +If `LC_ALL` is non-empty, Bash uses that locale rather than selecting another: +Unicode names fail with an error if the selected locale is not usable for UTF-8 +names. With `LC_ALL` unset or empty, Bash selects an installed UTF-8 locale even +when `LANG` or `LC_CTYPE` names a non-UTF-8 locale. ASCII capitals are lowercased; non-ASCII letter casing is preserved across the script variants. diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index ad944a475b..f3b9ff8176 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -152,7 +152,11 @@ spec_prefix_exists() { # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. UNICODE_LOCALE="" -for candidate in C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_ALL:-${LC_CTYPE:-${LANG:-}}}"; do +locale_candidates=(C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_CTYPE:-${LANG:-}}") +if [ -n "${LC_ALL:-}" ]; then + locale_candidates=("$LC_ALL") +fi +for candidate in "${locale_candidates[@]}"; do if [ -n "$candidate" ] && [ "$(printf 'é。' | LC_ALL="$candidate" sed 's/[^[:alnum:]]/-/g' 2>/dev/null)" = 'é-' ]; then UNICODE_LOCALE="$candidate" break @@ -162,7 +166,11 @@ done if [ -z "$UNICODE_LOCALE" ]; then UNICODE_LOCALE=C if printf '%s' "${SHORT_NAME:-$FEATURE_DESCRIPTION}" | LC_ALL=C grep -q '[^ -~]'; then - echo "Error: A UTF-8 locale is required to create a Unicode feature name" >&2 + if [ -n "${LC_ALL:-}" ]; then + echo "Error: A UTF-8 locale is required to create a Unicode feature name; LC_ALL=$LC_ALL is not usable" >&2 + else + echo "Error: A UTF-8 locale is required to create a Unicode feature name" >&2 + fi exit 1 fi fi @@ -279,7 +287,7 @@ generate_branch_name() { [ -z "$word" ] && continue # Retain non-ASCII words even when shorter than three characters. - if ! echo "$word" | grep -qiE "$stop_words"; then + if ! printf '%s\n' "$word" | LC_ALL=C grep -qE "$stop_words"; then if [ ${#word} -ge 3 ] || printf '%s' "$word" | LC_ALL=C grep -q '[^ -~]'; then meaningful_words+=("$word") # Keep short words that appear as an uppercase acronym in the original. diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index e60d850756..eda1ad8a27 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -217,6 +217,32 @@ def test_bash_reports_missing_utf8_locale_for_unicode_only( assert not (repo / "specs").exists() +@requires_bash +@pytest.mark.parametrize("locale_name", ["C", "POSIX"]) +def test_bash_respects_explicit_non_utf8_lc_all(repo: Path, locale_name: str) -> None: + env = clean_env() + env["LC_ALL"] = locale_name + env["LANG"] = "C.UTF-8" + + ascii_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "Add user authentication"), + repo, + env, + ) + assert ascii_result.returncode == 0, ascii_result.stderr + assert json_stdout(ascii_result)["BRANCH_NAME"] == "001-user-authentication" + + for args in (("添加用户",), ("--short-name", "用户", "Add users")): + unicode_result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", *args), repo, env + ) + assert unicode_result.returncode == 1 + assert unicode_result.stdout == "" + assert "A UTF-8 locale is required" in unicode_result.stderr + assert "LC_ALL" in unicode_result.stderr + assert not (repo / "specs").exists() + + @requires_bash def test_bash_requires_python_only_for_unicode_names( repo: Path, tmp_path: Path @@ -381,6 +407,54 @@ def test_python_branch_name_generation_matches_bash( assert json_stdout(py) == json_stdout(bash) +@requires_bash +@pytest.mark.parametrize( + ("description", "expected"), + [("ſet account", "001-ſet-account"), ("Set account", "001-account")], +) +def test_bash_stop_words_match_only_ascii_words( + repo: Path, description: str, expected: str +) -> None: + bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) + py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", description), repo) + + assert bash.returncode == py.returncode == 0 + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == expected + + +@requires_bash +def test_bash_stop_words_ignore_locale_case_folding(repo: Path, tmp_path: Path) -> None: + real_grep = shutil.which("grep") + assert real_grep is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + grep_shim = shim_dir / "grep" + grep_cmd = shlex.quote(_bash_posix_path(Path(real_grep))) + grep_shim.write_text( + "#!/bin/sh\n" + 'if [ "$1" = "-qiE" ]; then\n' + " IFS= read -r word\n" + ' [ "$word" = "ſet" ] && exit 0\n' + f' printf "%s\\n" "$word" | {grep_cmd} "$@"\n' + " exit $?\n" + "fi\n" + f'exec {grep_cmd} "$@"\n', + encoding="utf-8", + newline="\n", + ) + grep_shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + bash = run(bash_cmd(repo, SCRIPT, "--json", "--dry-run", "ſet account"), repo, env) + py = run(py_cmd(repo, SCRIPT, "--json", "--dry-run", "ſet account"), repo, env) + + assert bash.returncode == py.returncode == 0 + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == "001-ſet-account" + + @requires_bash @pytest.mark.skipif(not HAS_POWERSHELL, reason="no PowerShell available") @pytest.mark.parametrize( @@ -1489,7 +1563,8 @@ def test_no_ascii_word_description_matches_across_twins( ps_repo = _setup_repo(tmp_path, "s") env = clean_env() - env["LC_ALL"] = env["LANG"] = "C" + env.pop("LC_ALL", None) + env["LANG"] = "C" bash = run( bash_cmd(bash_repo, SCRIPT, "--json", "--dry-run", description), bash_repo, env ) From 7e46f9f37d73e94b2a13b46f17e6c44c39142f5b Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:02:50 -0500 Subject: [PATCH 7/9] fix: sanitize ASCII controls and bound Unicode truncation Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/reference/core.md | 1 + scripts/bash/create-new-feature.sh | 27 ++++-- .../test_create_new_feature_python_parity.py | 96 ++++++++++++++++++- 3 files changed, 116 insertions(+), 8 deletions(-) diff --git a/docs/reference/core.md b/docs/reference/core.md index 85ce7881ae..d94bc04621 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -80,6 +80,7 @@ The Python helper also accepts `--short-name`; the PowerShell helper uses `-ShortName`. A supplied short name is cleaned by the same rules, so it must contain at least one letter or digit. For non-ASCII names, the Bash helper needs an installed UTF-8 locale and a Python 3 interpreter for Unicode classification. +ASCII input, including tabs and newlines, is sanitized without either requirement. If `LC_ALL` is non-empty, Bash uses that locale rather than selecting another: Unicode names fail with an error if the selected locale is not usable for UTF-8 names. With `LC_ALL` unset or empty, Bash selects an installed UTF-8 locale even diff --git a/scripts/bash/create-new-feature.sh b/scripts/bash/create-new-feature.sh index f3b9ff8176..3ce2f9011c 100644 --- a/scripts/bash/create-new-feature.sh +++ b/scripts/bash/create-new-feature.sh @@ -151,6 +151,10 @@ spec_prefix_exists() { # * `--*` instead of the GNU-only `\+`, which POSIX/BSD sed reads as a literal # '+', leaving repeated separators uncollapsed on macOS. # * printf instead of echo, so a name of "-n"/"-e"/"-E" is text, not options. +contains_non_ascii() { + LC_ALL=C grep -q '[^[:print:][:cntrl:]]' +} + UNICODE_LOCALE="" locale_candidates=(C.UTF-8 C.utf8 en_US.UTF-8 en_US.utf8 "${LC_CTYPE:-${LANG:-}}") if [ -n "${LC_ALL:-}" ]; then @@ -165,7 +169,7 @@ done if [ -z "$UNICODE_LOCALE" ]; then UNICODE_LOCALE=C - if printf '%s' "${SHORT_NAME:-$FEATURE_DESCRIPTION}" | LC_ALL=C grep -q '[^ -~]'; then + if printf '%s' "${SHORT_NAME:-$FEATURE_DESCRIPTION}" | contains_non_ascii; then if [ -n "${LC_ALL:-}" ]; then echo "Error: A UTF-8 locale is required to create a Unicode feature name; LC_ALL=$LC_ALL is not usable" >&2 else @@ -176,9 +180,9 @@ if [ -z "$UNICODE_LOCALE" ]; then fi unicode_words() { - local name="$1" + local name="${1//$'\n'/ }" local separator="$2" - if printf '%s' "$name" | LC_ALL=C grep -q '[^ -~]'; then + if printf '%s' "$name" | contains_non_ascii; then local -a python_cmd=() local override="${SPECKIT_PYTHON_EXECUTABLE:-${SPECKIT_PYTHON:-}}" if [ -n "$override" ] && command -v "$override" >/dev/null 2>&1 && @@ -231,10 +235,19 @@ fit_branch_name() { local max_suffix_length=$((MAX_BRANCH_LENGTH - prefix_length)) local truncated_suffix local -x LC_ALL="$UNICODE_LOCALE" - truncated_suffix="${branch_suffix:0:$max_suffix_length}" - while [ "$(branch_byte_count "$truncated_suffix")" -gt "$max_suffix_length" ]; do - truncated_suffix="${truncated_suffix%?}" + local low=0 high=${#branch_suffix} mid + if (( high > max_suffix_length )); then + high=$max_suffix_length + fi + while (( low < high )); do + mid=$(((low + high + 1) / 2)) + if [ "$(branch_byte_count "${branch_suffix:0:$mid}")" -le "$max_suffix_length" ]; then + low=$mid + else + high=$((mid - 1)) + fi done + truncated_suffix="${branch_suffix:0:$low}" truncated_suffix="${truncated_suffix%-}" branch_name="${feature_num}-${truncated_suffix}" fi @@ -288,7 +301,7 @@ generate_branch_name() { # Retain non-ASCII words even when shorter than three characters. if ! printf '%s\n' "$word" | LC_ALL=C grep -qE "$stop_words"; then - if [ ${#word} -ge 3 ] || printf '%s' "$word" | LC_ALL=C grep -q '[^ -~]'; then + if [ ${#word} -ge 3 ] || printf '%s' "$word" | contains_non_ascii; then meaningful_words+=("$word") # Keep short words that appear as an uppercase acronym in the original. # Uppercase via tr and match with grep -w (both portable) rather than diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index eda1ad8a27..d76ff25227 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -17,6 +17,7 @@ from tests.conftest import requires_bash from tests.parity_helpers import ( HAS_POWERSHELL, + WINDOWS_POWERSHELL, _bash_posix_path, bash_cmd, break_wrap_layer, @@ -243,6 +244,42 @@ def test_bash_respects_explicit_non_utf8_lc_all(repo: Path, locale_name: str) -> assert not (repo / "specs").exists() +@requires_bash +@pytest.mark.parametrize( + "short_name", ["foo\tbar", "foo\nbar"], ids=["tab", "newline"] +) +def test_bash_ascii_controls_need_no_utf8_locale_or_python( + repo: Path, tmp_path: Path, short_name: str +) -> None: + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + for name in ("python3", "python", "py"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + + for lc_all in ("C", None): + env = clean_env() + if lc_all is not None: + env["LC_ALL"] = lc_all + else: + env.pop("LC_ALL", None) + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + bash = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", short_name, "x"), + repo, + env, + ) + py = run( + py_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", short_name, "x"), + repo, + env, + ) + assert bash.returncode == py.returncode == 0, bash.stderr + assert json_stdout(bash) == json_stdout(py) + assert json_stdout(bash)["BRANCH_NAME"] == "001-foo-bar" + + @requires_bash def test_bash_requires_python_only_for_unicode_names( repo: Path, tmp_path: Path @@ -1582,6 +1619,32 @@ def test_no_ascii_word_description_matches_across_twins( assert names == {expected}, names +@pytest.mark.skipif(WINDOWS_POWERSHELL is None, reason="Windows PowerShell unavailable") +def test_windows_powershell_51_preserves_unicode_and_utf8_limit(repo: Path) -> None: + assert WINDOWS_POWERSHELL is not None + for short_name, expected in ( + ("添加用户", "001-添加用户"), + ("𠀀" * 240, "001-" + "𠀀" * 60), + ): + result = run( + [ + WINDOWS_POWERSHELL, + "-NoProfile", + "-File", + str(repo / ".specify/scripts/powershell/create-new-feature.ps1"), + "-Json", + "-DryRun", + "-ShortName", + short_name, + "x", + ], + repo, + ) + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + assert len(expected.encode("utf-8")) <= 244 + + @requires_bash @pytest.mark.parametrize("variant", ["bash", "python", "powershell"]) @pytest.mark.parametrize( @@ -1589,8 +1652,9 @@ def test_no_ascii_word_description_matches_across_twins( [ ("客" * 100, "客" * 80), ("a" + "客" * 80, "a" + "客" * 79), + ("𠀀" * 240, "𠀀" * 60), ], - ids=["exact_boundary", "partial_codepoint"], + ids=["exact_boundary", "partial_codepoint", "four_byte_long_suffix"], ) def test_unicode_branch_name_fits_244_bytes( repo: Path, variant: str, short_name: str, expected_suffix: str @@ -1615,3 +1679,33 @@ def test_unicode_branch_name_fits_244_bytes( assert subprocess.run( ["git", "check-ref-format", "--branch", branch], capture_output=True ).returncode == 0 + + +@requires_bash +def test_bash_truncation_uses_bounded_byte_checks(repo: Path, tmp_path: Path) -> None: + real_wc = shutil.which("wc") + assert real_wc is not None + shim_dir = tmp_path / "bin" + shim_dir.mkdir() + count_file = tmp_path / "wc-count" + shim = shim_dir / "wc" + shim.write_text( + "#!/bin/sh\n" + f"printf . >> {shlex.quote(_bash_posix_path(count_file))}\n" + f"exec {shlex.quote(_bash_posix_path(Path(real_wc)))} \"$@\"\n", + encoding="utf-8", + newline="\n", + ) + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + + result = run( + bash_cmd(repo, SCRIPT, "--json", "--dry-run", "--short-name", "𠀀" * 240, "x"), + repo, + env, + ) + + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == "001-" + "𠀀" * 60 + assert count_file.read_text(encoding="utf-8").count(".") <= 16 From 3f1c32f4216721a24df993f107ca8069b18f9d3f Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:38:23 -0500 Subject: [PATCH 8/9] fix: read Unicode feature state as UTF-8 in Bash Use case-sensitive PowerShell stop-word matching after ASCII normalization. Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- scripts/bash/common.sh | 2 +- scripts/powershell/create-new-feature.ps1 | 2 +- .../test_create_new_feature_python_parity.py | 73 +++++++++++++++++++ 3 files changed, 75 insertions(+), 2 deletions(-) diff --git a/scripts/bash/common.sh b/scripts/bash/common.sh index 7b1516d29c..ff18cba4d9 100644 --- a/scripts/bash/common.sh +++ b/scripts/bash/common.sh @@ -112,7 +112,7 @@ read_feature_json_feature_directory() { fi if [[ -z "$_fd" ]] && command -v python3 >/dev/null 2>&1; then # Use Python so pretty-printed/multi-line JSON still parses correctly. - if ! _fd=$(python3 -c "import json,sys; d=json.load(open(sys.argv[1])); v=d.get('feature_directory'); print(v if v else '')" "$fj" 2>/dev/null); then + if ! _fd=$(python3 -c "import json,sys; d=json.load(open(sys.argv[1], encoding='utf-8')); v=d.get('feature_directory'); print(v if v else '')" "$fj" 2>/dev/null); then _fd='' fi fi diff --git a/scripts/powershell/create-new-feature.ps1 b/scripts/powershell/create-new-feature.ps1 index 8688332375..cd8a2d2b3c 100644 --- a/scripts/powershell/create-new-feature.ps1 +++ b/scripts/powershell/create-new-feature.ps1 @@ -167,7 +167,7 @@ function Get-BranchName { $meaningfulWords = @() foreach ($word in $words) { # Skip stop words - if ($stopWords -contains $word) { continue } + if ($stopWords -ccontains $word) { continue } # Keep Unicode words even when short; ASCII words still need three # characters or an uppercase acronym in the original. diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index d76ff25227..14b395e039 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -26,6 +26,7 @@ install_composition_stack, install_scripts, json_stdout, + make_python3_path_shim, make_repo, make_yaml_less_venv, normalize_repo_paths, @@ -458,6 +459,10 @@ def test_bash_stop_words_match_only_ascii_words( assert bash.returncode == py.returncode == 0 assert json_stdout(bash) == json_stdout(py) assert json_stdout(bash)["BRANCH_NAME"] == expected + if HAS_POWERSHELL: + ps = run(ps_cmd(repo, SCRIPT, "-Json", "-DryRun", description), repo) + assert ps.returncode == 0, ps.stderr + assert json_stdout(ps) == json_stdout(py) @requires_bash @@ -993,6 +998,55 @@ def test_python_persists_relative_feature_json(repo: Path) -> None: assert feature_json == f'{{"feature_directory":"specs/{branch}"}}\n' +@requires_bash +def test_bash_reads_unicode_feature_state_without_jq_under_legacy_encoding( + repo: Path, tmp_path: Path +) -> None: + created = run(bash_cmd(repo, SCRIPT, "--json", "添加用户"), repo) + assert created.returncode == 0, created.stderr + assert json_stdout(created)["BRANCH_NAME"] == "001-添加用户" + assert (repo / "specs/001-添加用户/spec.md").is_file() + + shim_dir = make_python3_path_shim(tmp_path / "bin") + (shim_dir / "sitecustomize.py").write_text( + "import builtins\n" + "_open = builtins.open\n" + "def legacy_open(file, *args, **kwargs):\n" + " if str(file).endswith('feature.json') and 'encoding' not in kwargs:\n" + " kwargs['encoding'] = 'cp1252'\n" + " return _open(file, *args, **kwargs)\n" + "builtins.open = legacy_open\n", + encoding="utf-8", + ) + for name in ("jq", "grep"): + shim = shim_dir / name + shim.write_text("#!/bin/sh\nexit 1\n", encoding="utf-8", newline="\n") + shim.chmod(0o755) + env = clean_env() + env["PATH"] = f"{shim_dir}{os.pathsep}{env['PATH']}" + env["PYTHONPATH"] = str(shim_dir) + env["PYTHONIOENCODING"] = "utf-8" + common = repo / ".specify/scripts/bash/common.sh" + + resolved = run( + [ + "bash", + "-c", + 'source "$1"; read_feature_json_feature_directory "$2"; printf "\\n"; get_feature_paths --no-persist', + "bash", + str(common), + str(repo), + ], + repo, + env, + ) + + assert resolved.returncode == 0, resolved.stderr + assert resolved.stdout.splitlines()[0] == "specs/001-添加用户" + assert "CURRENT_BRANCH=" in resolved.stdout + assert (repo / resolved.stdout.splitlines()[0] / "spec.md").is_file() + + def test_persist_feature_json_avoids_platform_newline_translation( tmp_path: Path, monkeypatch ) -> None: @@ -1644,6 +1698,25 @@ def test_windows_powershell_51_preserves_unicode_and_utf8_limit(repo: Path) -> N assert json_stdout(result)["BRANCH_NAME"] == expected assert len(expected.encode("utf-8")) <= 244 + for description, expected in ( + ("ſet account", "001-ſet-account"), + ("Set account", "001-account"), + ): + result = run( + [ + WINDOWS_POWERSHELL, + "-NoProfile", + "-File", + str(repo / ".specify/scripts/powershell/create-new-feature.ps1"), + "-Json", + "-DryRun", + description, + ], + repo, + ) + assert result.returncode == 0, result.stderr + assert json_stdout(result)["BRANCH_NAME"] == expected + @requires_bash @pytest.mark.parametrize("variant", ["bash", "python", "powershell"]) From 846e20ade544b77892827954cedf31699e772d9b Mon Sep 17 00:00:00 2001 From: Manfred Riem <15701806+mnriem@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:51:48 -0500 Subject: [PATCH 9/9] test: keep Bash feature-state regression portable Assisted-by: GitHub Copilot (model: GPT-6 Sol, autonomous) Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- tests/test_create_new_feature_python_parity.py | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/tests/test_create_new_feature_python_parity.py b/tests/test_create_new_feature_python_parity.py index 14b395e039..177bdea30d 100644 --- a/tests/test_create_new_feature_python_parity.py +++ b/tests/test_create_new_feature_python_parity.py @@ -1032,19 +1032,24 @@ def test_bash_reads_unicode_feature_state_without_jq_under_legacy_encoding( [ "bash", "-c", - 'source "$1"; read_feature_json_feature_directory "$2"; printf "\\n"; get_feature_paths --no-persist', + ( + 'source "$1"; paths=$(get_feature_paths --no-persist) || exit 1; ' + 'printf -v expected "FEATURE_DIR=%q" "$3"; ' + '[[ "$paths" == *"$expected"* ]] || exit 1; ' + 'read_feature_json_feature_directory "$2"' + ), "bash", str(common), str(repo), + str(repo / "specs/001-添加用户"), ], repo, env, ) assert resolved.returncode == 0, resolved.stderr - assert resolved.stdout.splitlines()[0] == "specs/001-添加用户" - assert "CURRENT_BRANCH=" in resolved.stdout - assert (repo / resolved.stdout.splitlines()[0] / "spec.md").is_file() + assert resolved.stdout == "specs/001-添加用户" + assert (repo / resolved.stdout / "spec.md").is_file() def test_persist_feature_json_avoids_platform_newline_translation(