diff --git a/src/iniconfig/_parse.py b/src/iniconfig/_parse.py index 57b9b44..e7a1a8c 100644 --- a/src/iniconfig/_parse.py +++ b/src/iniconfig/_parse.py @@ -1,9 +1,11 @@ from collections.abc import Mapping +from typing import Final from typing import NamedTuple from .exceptions import ParseError COMMENTCHARS = "#;" +UTF8_BOM: Final = "\N{BYTE ORDER MARK}" class ParsedLine(NamedTuple): @@ -34,6 +36,8 @@ def parse_ini_data( - sections_data: mapping of section -> {name -> value} - sources: mapping of (section, name) -> line number """ + data = data.removeprefix(UTF8_BOM) + tokens = parse_lines( path, data.splitlines(True), diff --git a/testing/test_iniconfig.py b/testing/test_iniconfig.py index 85193c5..6162256 100644 --- a/testing/test_iniconfig.py +++ b/testing/test_iniconfig.py @@ -412,3 +412,23 @@ def test_unicode_whitespace_in_key_names() -> None: ) assert "key" in config["section"] assert config["section"]["key"] == "value" + + +def test_utf8_bom_file(tmp_path: Path) -> None: + """Files saved with a UTF-8 BOM (common on Windows editors) must parse.""" + path = tmp_path / "bom.ini" + path.write_bytes(b"\xef\xbb\xbf[section]\nkey = value\n") + config = IniConfig(path) + assert config["section"]["key"] == "value" + + +def test_utf8_bom_in_data_string() -> None: + config = IniConfig("x.ini", data="\ufeff[section]\nkey = value\n") + assert config["section"]["key"] == "value" + + +def test_parse_utf8_bom_file(tmp_path: Path) -> None: + path = tmp_path / "bom.ini" + path.write_bytes(b"\xef\xbb\xbf[section]\nkey = value\n") + config = IniConfig.parse(path) + assert config["section"]["key"] == "value"