From 21d6602969dee9fd6974280bd857d9048f428b1c Mon Sep 17 00:00:00 2001 From: Jahnvi Thakkar Date: Wed, 30 Sep 2026 17:48:19 +0530 Subject: [PATCH 1/2] FIX: Warn when setencoding settings cannot be applied Preserve wide-character binding while warning about ignored encoding requests. Align native encoding gates and document and test the effective contract. Refs #825; AB#48882 Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- README.md | 17 ++ mssql_python/connection.py | 46 ++++- mssql_python/constants.py | 1 + mssql_python/pybind/ddbc_bindings.cpp | 21 +- tests/test_013_encoding_decoding.py | 282 +++++++++++++------------- 5 files changed, 209 insertions(+), 158 deletions(-) diff --git a/README.md b/README.md index 61954e9f7..f7efa6903 100644 --- a/README.md +++ b/README.md @@ -151,6 +151,23 @@ By adhering to the DB API 2.0 specification, the mssql-python module ensures com The driver offers a suite of Pythonic enhancements that streamline database interactions, making it easier for developers to execute queries, manage connections, and handle data more efficiently. +### Text encoding + +SQL statements and Python `str` parameters always use UTF-16LE. Text parameters are +bound as ODBC `SQL_C_WCHAR` on every supported platform, for both `execute()` and +`executemany()`, including calls that use `setinputsizes()`. Declaring a `VARCHAR` +SQL type does not switch to narrow C buffers; SQL Server performs the conversion +to the destination column's character set. + +`Connection.setencoding()` retains requested settings for compatibility, but does +not change statement encoding or parameter binding. Requests other than +`encoding="utf-16le", ctype=SQL_WCHAR` emit `UserWarning`, including an explicitly +requested or automatically selected `SQL_CHAR`. `getencoding()` returns the +requested settings, not the effective binding. For example, requesting ASCII does +not cause non-ASCII parameters to raise encoding errors. Use `setencoding()` with +no arguments to restore the supported defaults. `setdecoding()` independently +controls how result data is read. + ## Getting Started Examples Connect to SQL Server and execute a simple query: diff --git a/mssql_python/connection.py b/mssql_python/connection.py index 1984f4979..496ada752 100644 --- a/mssql_python/connection.py +++ b/mssql_python/connection.py @@ -1183,15 +1183,21 @@ def setautocommit(self, value: bool = False) -> None: def setencoding(self, encoding: Optional[str] = None, ctype: Optional[int] = None) -> None: """ - Sets the text encoding for SQL statements and text parameters. + Records the requested text encoding settings for compatibility. - Since Python 3 only has str (which is Unicode), this method configures - how text is encoded when sending to the database. + SQL statements and str parameters are always sent as UTF-16LE; text + parameters are bound as SQL_C_WCHAR on every platform. This applies to + execute() and executemany(), with or without setinputsizes(). This method + does not change that behavior or enforce the requested codec. + + Requests other than UTF-16LE with SQL_WCHAR emit UserWarning. The requested + settings are still returned by getencoding(), not the effective binding. + Use setdecoding() separately to configure how results are read. Args: - encoding (str, optional): The encoding to use. This must be a valid Python + encoding (str, optional): The requested encoding. This must be a valid Python encoding that converts text to bytes. If None, defaults to 'utf-16le'. - ctype (int, optional): The C data type to use when passing data: + ctype (int, optional): The requested C data type: SQL_CHAR or SQL_WCHAR. If not provided, SQL_WCHAR is used for UTF-16 variants (see UTF16_ENCODINGS constant). SQL_CHAR is used for all other encodings. @@ -1203,12 +1209,15 @@ def setencoding(self, encoding: Optional[str] = None, ctype: Optional[int] = Non ProgrammingError: If the encoding is not valid or not supported. InterfaceError: If the connection is closed. + Warns: + UserWarning: If the requested encoding or ctype cannot be honored. + Example: - # For databases that only communicate with UTF-8 - cnxn.setencoding(encoding='utf-8') + # Restore the supported default. + cnxn.setencoding() - # For explicitly using SQL_CHAR - cnxn.setencoding(encoding='utf-8', ctype=mssql_python.SQL_CHAR) + # Warns: parameters still use UTF-16LE / SQL_C_WCHAR. + cnxn.setencoding(encoding='cp1252', ctype=mssql_python.SQL_CHAR) """ logger.debug( "setencoding: Configuring encoding=%s, ctype=%s", @@ -1276,20 +1285,35 @@ def setencoding(self, encoding: Optional[str] = None, ctype: Optional[int] = Non if ctype == ConstantsDDBC.SQL_WCHAR.value: _validate_utf16_wchar_compatibility(encoding, ctype, "SQL_WCHAR") + if encoding != "utf-16le" or ctype != ConstantsDDBC.SQL_WCHAR.value: + warnings.warn( + "setencoding() does not change SQL statement encoding or text parameter binding: " + "statements and str parameters always use UTF-16LE, and text parameters are " + "bound as SQL_C_WCHAR. The requested settings are retained by getencoding() " + "for compatibility but are not applied. Use setencoding() with no arguments " + "to restore the supported defaults.", + UserWarning, + stacklevel=2, + ) + # Store the encoding settings (thread-safe with lock) with self._encoding_lock: self._encoding_settings = {"encoding": encoding, "ctype": ctype} # Log with sanitized values for security logger.info( - "Text encoding set to %s with ctype %s", + "Requested text encoding stored as %s with ctype %s", sanitize_user_input(encoding), sanitize_user_input(str(ctype)), ) def getencoding(self) -> Dict[str, Union[str, int]]: """ - Gets the current text encoding settings (thread-safe). + Gets the requested text encoding settings (thread-safe). + + These settings are retained for compatibility. They do not describe the + effective binding: SQL statements and str parameters always use UTF-16LE, + and text parameters are bound as SQL_C_WCHAR. Returns: dict: A dictionary containing 'encoding' and 'ctype' keys. diff --git a/mssql_python/constants.py b/mssql_python/constants.py index 54a51b9ce..0538d0934 100644 --- a/mssql_python/constants.py +++ b/mssql_python/constants.py @@ -75,6 +75,7 @@ class ConstantsDDBC(Enum): SQL_C_VARBINARY = -3 SQL_C_LONGVARBINARY = -4 SQL_C_LONGVARCHAR = -1 + # Legacy alias: text parameters bind as ODBC SQL_C_WCHAR (-8), not SQL_C_CHAR (1). SQL_C_CHAR = -8 SQL_C_NUMERIC = 2 SQL_C_DECIMAL = 3 diff --git a/mssql_python/pybind/ddbc_bindings.cpp b/mssql_python/pybind/ddbc_bindings.cpp index f410631d4..4e4523955 100644 --- a/mssql_python/pybind/ddbc_bindings.cpp +++ b/mssql_python/pybind/ddbc_bindings.cpp @@ -2136,14 +2136,10 @@ SQLRETURN SQLExecute_wrap(const SqlHandlePtr statementHandle, (SQLPOINTER)SQL_CONCUR_READ_ONLY, 0); } - // The encoding-settings dict has the form {"encoding": str, "ctype": int}. - // Note: the Python layer's SQL_C_CHAR constant is numerically -8, the same - // as ODBC's SQL_C_WCHAR. As a result, the only path that genuinely uses - // byte-level character encoding is when the user explicitly opts in via - // setencoding(..., ctype=mssql_python.SQL_CHAR) (which sends ctype=1, the - // real ODBC SQL_CHAR). We default to utf-8 and only honor the dict's - // encoding when ctype == 1 (real ODBC SQL_CHAR). Otherwise the user's - // "encoding" value is meant for the wide-char path and we leave it alone. + // This codec only applies to parameters already typed as real SQL_C_CHAR (1). + // Public text parameter detection uses SQL_C_WCHAR (-8), including the + // Python layer's legacy SQL_C_CHAR alias. setencoding() does not change + // paramCType and warns when the requested settings cannot be applied. std::string charEncoding = "utf-8"; if (encoding_settings.contains("ctype") && encoding_settings.contains("encoding")) { int ctype = encoding_settings["ctype"].cast(); @@ -2959,10 +2955,13 @@ SQLRETURN SQLExecuteMany_wrap(const SqlHandlePtr statementHandle, const std::u16 } LOG("SQLExecuteMany: Parameter analysis - hasDAE=%s", hasDAE ? "true" : "false"); - // Extract char encoding from encodingSettings dictionary + // Match SQLExecute_wrap: a wide-char codec must never encode narrow buffers. std::string charEncoding = "utf-8"; // default - if (encodingSettings.contains("encoding")) { - charEncoding = encodingSettings["encoding"].cast(); + if (encodingSettings.contains("ctype") && encodingSettings.contains("encoding")) { + int ctype = encodingSettings["ctype"].cast(); + if (ctype == SQL_C_CHAR) { + charEncoding = encodingSettings["encoding"].cast(); + } } if (!hasDAE) { diff --git a/tests/test_013_encoding_decoding.py b/tests/test_013_encoding_decoding.py index a03e79db7..d1912d3da 100644 --- a/tests/test_013_encoding_decoding.py +++ b/tests/test_013_encoding_decoding.py @@ -105,6 +105,7 @@ from mssql_python import db_connection import pytest import sys +import warnings import mssql_python from mssql_python import connect, SQL_CHAR, SQL_WCHAR, SQL_WMETADATA from mssql_python.exceptions import ( @@ -125,6 +126,58 @@ def test_setencoding_default_settings(db_connection): assert settings["ctype"] == -8, "Default ctype should be SQL_WCHAR (-8)" +@pytest.mark.parametrize( + "encoding, ctype, expected_encoding, expected_ctype", + [ + ("cp1252", SQL_CHAR, "cp1252", SQL_CHAR), + ("cp1252", None, "cp1252", SQL_CHAR), + ("ascii", SQL_CHAR, "ascii", SQL_CHAR), + ("UTF-8", None, "utf-8", SQL_CHAR), + ("shift_jis", SQL_CHAR, "shift_jis", SQL_CHAR), + ("utf-16le", SQL_CHAR, "utf-16le", SQL_CHAR), + (None, SQL_CHAR, "utf-16le", SQL_CHAR), + ("utf-16be", SQL_WCHAR, "utf-16be", SQL_WCHAR), + ("utf-16be", None, "utf-16be", SQL_WCHAR), + ], +) +def test_setencoding_warns_for_unsupported_binding( + conn_str, encoding, ctype, expected_encoding, expected_ctype +): + with connect(conn_str) as conn: + with pytest.warns(UserWarning, match="UTF-16LE.*SQL_C_WCHAR") as caught: + conn.setencoding(encoding, ctype) + + assert len(caught) == 1 + assert caught[0].filename == __file__ + assert conn.getencoding() == { + "encoding": expected_encoding, + "ctype": expected_ctype, + } + + +@pytest.mark.parametrize( + "encoding, ctype", + [(None, None), ("utf-16le", None), ("utf-16le", SQL_WCHAR), ("UTF-16LE", SQL_WCHAR)], +) +def test_setencoding_supported_binding_is_silent(conn_str, encoding, ctype): + with connect(conn_str) as conn: + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + conn.setencoding(encoding, ctype) + assert not caught + assert conn.getencoding() == {"encoding": "utf-16le", "ctype": SQL_WCHAR} + + +def test_setencoding_warning_as_error_preserves_settings(conn_str): + with connect(conn_str) as conn: + original = conn.getencoding() + with warnings.catch_warnings(): + warnings.simplefilter("error", UserWarning) + with pytest.raises(UserWarning, match="UTF-16LE.*SQL_C_WCHAR"): + conn.setencoding("ascii", SQL_CHAR) + assert conn.getencoding() == original + + def test_setencoding_basic_functionality(db_connection): """Test basic setencoding functionality.""" # Test setting UTF-8 encoding @@ -3237,33 +3290,6 @@ def test_big5_encoding_chinese_traditional(db_connection): cursor.close() -def test_shift_jis_encoding_japanese(db_connection): - """Test Shift-JIS encoding for Japanese characters.""" - db_connection.setencoding(encoding="shift_jis", ctype=SQL_CHAR) - db_connection.setdecoding(SQL_CHAR, encoding="shift_jis", ctype=SQL_CHAR) - - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_sjis (id INT, data VARCHAR(200))") - - japanese_tests = [ - ("こんにちは", "Hello"), - ("東京", "Tokyo"), - ] - - for japanese_text, meaning in japanese_tests: - if is_encoding_compatible_with_data("shift_jis", japanese_text): - cursor.execute("DELETE FROM #test_sjis") - cursor.execute("INSERT INTO #test_sjis VALUES (?, ?)", 1, japanese_text) - cursor.execute("SELECT data FROM #test_sjis WHERE id = 1") - result = cursor.fetchone() - else: - pass - - finally: - cursor.close() - - def test_euc_kr_encoding_korean(db_connection): """Test EUC-KR encoding for Korean characters.""" db_connection.setencoding(encoding="euc-kr", ctype=SQL_CHAR) @@ -5938,77 +5964,31 @@ def test_encoding_with_special_characters_in_sql_char(db_connection): cursor.close() -def test_encoding_error_propagation_in_bind_parameters(db_connection): - """Test encoding behavior with incompatible characters (strict mode in C++ layer).""" - # Set ASCII encoding - in strict mode, C++ layer catches encoding errors - db_connection.setencoding(encoding="ascii", ctype=mssql_python.SQL_CHAR) - - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_encode_fail (id INT, data VARCHAR(100))") - - # With ASCII encoding and non-ASCII characters, the C++ layer will: - # 1. Attempt to encode with Python's str.encode('ascii', 'strict') - # 2. Raise UnicodeEncodeError which gets caught and re-raised as RuntimeError - error_raised = False - try: - cursor.execute( - "INSERT INTO #test_encode_fail (id, data) VALUES (?, ?)", 1, "Unicode: 你好" - ) - except (UnicodeEncodeError, RuntimeError, Exception) as e: - error_raised = True - # Verify it's an encoding-related error - error_str = str(e).lower() - assert ( - "encode" in error_str - or "ascii" in error_str - or "unicode" in error_str - or "codec" in error_str - or "failed" in error_str - ) - - # If no error was raised, that's also acceptable behavior (data may be mangled) - # The key is that the C++ code path was exercised - if not error_raised: - # Verify the operation completed (even if data is mangled) - cursor.execute("SELECT COUNT(*) FROM #test_encode_fail") - count = cursor.fetchone()[0] - assert count >= 0 - - finally: - cursor.close() - - -def test_sql_c_char_encoding_failure(db_connection): - """Test encoding failure handling in C++ layer (lines 337-345).""" - # Set an encoding and then try to encode data that can't be represented - db_connection.setencoding(encoding="ascii", ctype=mssql_python.SQL_CHAR) - - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_encode_fail_cpp (id INT, data VARCHAR(100))") - - # Try to insert non-ASCII characters with ASCII encoding - # This should trigger the encoding error path (lines 337-345) - error_raised = False - try: - cursor.execute( - "INSERT INTO #test_encode_fail_cpp (id, data) VALUES (?, ?)", - 1, - "Non-ASCII: 你好世界", - ) - except (UnicodeEncodeError, RuntimeError, Exception) as e: - error_raised = True - error_msg = str(e).lower() - assert any(word in error_msg for word in ["encode", "ascii", "codec", "failed"]) - - # Error should be raised in strict mode - if not error_raised: - # Some implementations may handle this differently - pass - - finally: - cursor.close() +@pytest.mark.parametrize("encoding", ["ascii", "cp1252", "shift_jis"]) +@pytest.mark.parametrize("method", ["execute", "executemany"]) +@pytest.mark.parametrize("use_inputsizes", [False, True]) +@pytest.mark.parametrize("large", [False, True], ids=["inline", "streamed"]) +def test_setencoding_warning_preserves_unicode_binding( + conn_str, encoding, method, use_inputsizes, large +): + """Unsupported settings warn, but do not change existing Unicode binding.""" + text = "caf\u00e9 \u4f60\u597d" * (1000 if large else 1) + with connect(conn_str) as conn: + with conn.cursor() as cursor: + cursor.execute("CREATE TABLE #encoding_warning (data NVARCHAR(MAX))") + with pytest.warns(UserWarning, match="UTF-16LE.*SQL_C_WCHAR"): + conn.setencoding(encoding, SQL_CHAR) + if use_inputsizes: + sql_type = mssql_python.SQL_WLONGVARCHAR if large else mssql_python.SQL_WVARCHAR + cursor.setinputsizes([(sql_type, len(text), 0)]) + if method == "executemany": + cursor.executemany("INSERT INTO #encoding_warning VALUES (?)", [(text,), (text,)]) + expected = [text, text] + else: + cursor.execute("INSERT INTO #encoding_warning VALUES (?)", text) + expected = [text] + cursor.execute("SELECT data FROM #encoding_warning") + assert [row[0] for row in cursor.fetchall()] == expected def test_dae_sql_c_char_with_various_data_types(db_connection): @@ -6292,20 +6272,68 @@ def test_binary_lob_fetching(db_connection): cursor.close() -def test_cpp_bind_params_str_encoding(db_connection): - """str encoding with SQL_C_CHAR.""" - db_connection.setencoding(encoding="utf-8", ctype=mssql_python.SQL_CHAR) - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_cpp_str (data VARCHAR(50))") - # This hits: py::isinstance(param) == true - # and: param.attr("encode")(charEncoding, "strict") - # Note: VARCHAR stores in DB collation (Latin1), so we use ASCII-compatible chars - cursor.execute("INSERT INTO #test_cpp_str VALUES (?)", "Hello UTF-8 Test") - cursor.execute("SELECT data FROM #test_cpp_str") - assert cursor.fetchone()[0] == "Hello UTF-8 Test" - finally: - cursor.close() +@pytest.mark.parametrize("method", ["execute", "executemany"]) +@pytest.mark.parametrize("use_inputsizes", [False, True]) +def test_cp1252_setting_warns_and_preserves_varchar_binding(conn_str, method, use_inputsizes): + with connect(conn_str) as conn: + with conn.cursor() as cursor: + cursor.execute( + "CREATE TABLE #cp1252_warning " + "(data VARCHAR(50) COLLATE SQL_Latin1_General_CP1_CI_AS)" + ) + with pytest.warns(UserWarning, match="UTF-16LE.*SQL_C_WCHAR"): + conn.setencoding("cp1252", SQL_CHAR) + if use_inputsizes: + cursor.setinputsizes([(mssql_python.SQL_VARCHAR, 50, 0)]) + text = "caf\u00e9" + if method == "executemany": + cursor.executemany("INSERT INTO #cp1252_warning VALUES (?)", [(text,), (text,)]) + expected = [b"caf\xe9", b"caf\xe9"] + else: + cursor.execute("INSERT INTO #cp1252_warning VALUES (?)", text) + expected = [b"caf\xe9"] + cursor.execute("SELECT CONVERT(VARBINARY(50), data) FROM #cp1252_warning") + assert [row[0] for row in cursor.fetchall()] == expected + + +@pytest.mark.parametrize("method", ["execute", "executemany"]) +@pytest.mark.parametrize( + "settings, expected", + [ + ({}, b"AB"), + ({"encoding": "utf-16le"}, b"AB"), + ({"encoding": "utf-16le", "ctype": SQL_WCHAR}, b"AB"), + ({"encoding": "utf-16le", "ctype": SQL_CHAR}, b"A\x00B\x00"), + ], +) +def test_native_narrow_binding_encoding_gate(conn_str, method, settings, expected): + """Both native entry points apply a codec only for a narrow-ctype request.""" + ddbc = mssql_python.ddbc_bindings + with connect(conn_str) as conn: + with conn.cursor() as cursor: + cursor.execute("CREATE TABLE #native_encoding_gate (data VARCHAR(50))") + query = "INSERT INTO #native_encoding_gate VALUES (?)" + # Bypass public type detection to exercise real ODBC SQL_C_CHAR (1). + if method == "execute": + result = ddbc.DDBCSQLExecute( + cursor.hstmt, + query, + ["AB"], + [(mssql_python.SQL_VARCHAR, SQL_CHAR, 50, 0)], + [False], + True, + settings, + ) + else: + info = ddbc.ParamInfo() + info.paramSQLType = mssql_python.SQL_VARCHAR + info.paramCType = SQL_CHAR + info.columnSize = 50 + info.inputOutputType = 1 + result = ddbc.SQLExecuteMany(cursor.hstmt, query, [["AB"]], [info], 1, settings) + assert result == 0 + cursor.execute("SELECT CONVERT(VARBINARY(50), data) FROM #native_encoding_gate") + assert cursor.fetchone()[0] == expected def test_cpp_bind_params_bytes_encoding(db_connection): @@ -6814,9 +6842,9 @@ def test_big5_encoding_traditional_chinese(db_connection): def test_shift_jis_encoding_japanese(db_connection): - """Test Shift-JIS encoding/decoding round-trip with Japanese characters using NVARCHAR.""" - # Set encoding for INSERT (Shift-JIS) and decoding for SELECT (UTF-16LE from NVARCHAR) - db_connection.setencoding(encoding="shift_jis", ctype=SQL_CHAR) + """A Shift-JIS request warns; Japanese text still round-trips through UTF-16LE.""" + with pytest.warns(UserWarning, match="UTF-16LE.*SQL_C_WCHAR"): + db_connection.setencoding(encoding="shift_jis", ctype=SQL_CHAR) db_connection.setdecoding(SQL_WCHAR, encoding="utf-16le", ctype=SQL_WCHAR) cursor = db_connection.cursor() @@ -6826,7 +6854,6 @@ def test_shift_jis_encoding_japanese(db_connection): ) cursor.execute("CREATE TABLE #test_shift_jis (id INT, data NVARCHAR(200))") - # Japanese strings (Shift-JIS encoding) japanese_strings = [ "こんにちは", # Hello (Hiragana) "ありがとう", # Thank you (Hiragana) @@ -6839,27 +6866,10 @@ def test_shift_jis_encoding_japanese(db_connection): "データベース", # Database (Katakana) ] - inserted_indices = [] for i, text in enumerate(japanese_strings, 1): - try: - cursor.execute("INSERT INTO #test_shift_jis (id, data) VALUES (?, ?)", i, text) - inserted_indices.append(i - 1) - except Exception: - # Shift-JIS encoding might fail with VARCHAR - pass - - # If any data was inserted, verify round-trip integrity - if inserted_indices: - cursor.execute("SELECT id, data FROM #test_shift_jis ORDER BY id") - results = cursor.fetchall() - - for idx, (row_id, retrieved_text) in enumerate(results): - original_idx = inserted_indices[idx] - expected_text = japanese_strings[original_idx] - assert retrieved_text == expected_text, ( - f"Round-trip failed for Japanese Shift-JIS text at index {original_idx}: " - f"expected '{expected_text}', got '{retrieved_text}'" - ) + cursor.execute("INSERT INTO #test_shift_jis (id, data) VALUES (?, ?)", i, text) + cursor.execute("SELECT data FROM #test_shift_jis ORDER BY id") + assert [row[0] for row in cursor.fetchall()] == japanese_strings finally: cursor.close() From 8ed67c558b1f76e1b7ad0133c2c29c2d747618b8 Mon Sep 17 00:00:00 2001 From: Jahnvi Thakkar Date: Thu, 1 Oct 2026 16:54:20 +0530 Subject: [PATCH 2/2] FIX: Clarify encoding validation and strengthen streaming coverage Address review feedback by documenting validation errors before warnings and removing permissive ASCII DAE tests. Assert public executemany uses DDBCSQLExecute for streaming, with exact UTF-16LE data preservation and native bridge call counts. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- README.md | 16 ++--- mssql_python/connection.py | 15 +++-- tests/test_013_encoding_decoding.py | 98 +++++++++++++---------------- 3 files changed, 63 insertions(+), 66 deletions(-) diff --git a/README.md b/README.md index f7efa6903..40fc409e7 100644 --- a/README.md +++ b/README.md @@ -160,13 +160,15 @@ SQL type does not switch to narrow C buffers; SQL Server performs the conversion to the destination column's character set. `Connection.setencoding()` retains requested settings for compatibility, but does -not change statement encoding or parameter binding. Requests other than -`encoding="utf-16le", ctype=SQL_WCHAR` emit `UserWarning`, including an explicitly -requested or automatically selected `SQL_CHAR`. `getencoding()` returns the -requested settings, not the effective binding. For example, requesting ASCII does -not cause non-ASCII parameters to raise encoding errors. Use `setencoding()` with -no arguments to restore the supported defaults. `setdecoding()` independently -controls how result data is read. +not change statement encoding or parameter binding. Requests that pass validation +but differ from `encoding="utf-16le", ctype=SQL_WCHAR` emit `UserWarning`, including +an explicitly requested or automatically selected `SQL_CHAR`. Invalid codec names, +invalid ctypes, and incompatible combinations (such as UTF-8 with `SQL_WCHAR`) +raise `ProgrammingError` before any warning is emitted or settings are stored. +`getencoding()` returns the requested settings, not the effective binding. For +example, requesting ASCII with `SQL_CHAR` does not cause non-ASCII parameters to +raise encoding errors. Use `setencoding()` with no arguments to restore the +supported defaults. `setdecoding()` independently controls how result data is read. ## Getting Started Examples Connect to SQL Server and execute a simple query: diff --git a/mssql_python/connection.py b/mssql_python/connection.py index 496ada752..4d0b4cd72 100644 --- a/mssql_python/connection.py +++ b/mssql_python/connection.py @@ -1190,9 +1190,12 @@ def setencoding(self, encoding: Optional[str] = None, ctype: Optional[int] = Non execute() and executemany(), with or without setinputsizes(). This method does not change that behavior or enforce the requested codec. - Requests other than UTF-16LE with SQL_WCHAR emit UserWarning. The requested - settings are still returned by getencoding(), not the effective binding. - Use setdecoding() separately to configure how results are read. + Requests that pass validation but differ from UTF-16LE with SQL_WCHAR emit + UserWarning. Invalid codec names, invalid ctypes, and incompatible + combinations (such as UTF-8 with SQL_WCHAR) raise ProgrammingError before + any warning is emitted or settings are stored. Accepted settings are still + returned by getencoding(), not the effective binding. Use setdecoding() + separately to configure how results are read. Args: encoding (str, optional): The requested encoding. This must be a valid Python @@ -1206,11 +1209,13 @@ def setencoding(self, encoding: Optional[str] = None, ctype: Optional[int] = Non None Raises: - ProgrammingError: If the encoding is not valid or not supported. + ProgrammingError: If the encoding or ctype is invalid, or their + combination is incompatible. InterfaceError: If the connection is closed. Warns: - UserWarning: If the requested encoding or ctype cannot be honored. + UserWarning: If the request passes validation but its encoding or + ctype cannot be honored. Example: # Restore the supported default. diff --git a/tests/test_013_encoding_decoding.py b/tests/test_013_encoding_decoding.py index d1912d3da..5559aadcf 100644 --- a/tests/test_013_encoding_decoding.py +++ b/tests/test_013_encoding_decoding.py @@ -106,6 +106,7 @@ import pytest import sys import warnings +from unittest.mock import patch import mssql_python from mssql_python import connect, SQL_CHAR, SQL_WCHAR, SQL_WMETADATA from mssql_python.exceptions import ( @@ -178,6 +179,27 @@ def test_setencoding_warning_as_error_preserves_settings(conn_str): assert conn.getencoding() == original +@pytest.mark.parametrize( + "encoding, ctype, error", + [ + ("invalid-encoding-name", SQL_CHAR, "Unsupported encoding"), + ("utf-8", 999, "Invalid ctype"), + ("utf-8", SQL_WCHAR, "SQL_WCHAR only supports UTF-16 encodings"), + ("ascii", SQL_WCHAR, "SQL_WCHAR only supports UTF-16 encodings"), + ("utf-16", SQL_WCHAR, "Byte Order Mark not supported"), + ], +) +def test_setencoding_invalid_request_raises_without_warning(conn_str, encoding, ctype, error): + with connect(conn_str) as conn: + original = conn.getencoding() + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + with pytest.raises(ProgrammingError, match=error): + conn.setencoding(encoding, ctype) + assert not caught + assert conn.getencoding() == original + + def test_setencoding_basic_functionality(db_connection): """Test basic setencoding functionality.""" # Test setting UTF-8 encoding @@ -5973,6 +5995,7 @@ def test_setencoding_warning_preserves_unicode_binding( ): """Unsupported settings warn, but do not change existing Unicode binding.""" text = "caf\u00e9 \u4f60\u597d" * (1000 if large else 1) + ddbc = mssql_python.ddbc_bindings with connect(conn_str) as conn: with conn.cursor() as cursor: cursor.execute("CREATE TABLE #encoding_warning (data NVARCHAR(MAX))") @@ -5981,14 +6004,27 @@ def test_setencoding_warning_preserves_unicode_binding( if use_inputsizes: sql_type = mssql_python.SQL_WLONGVARCHAR if large else mssql_python.SQL_WVARCHAR cursor.setinputsizes([(sql_type, len(text), 0)]) - if method == "executemany": - cursor.executemany("INSERT INTO #encoding_warning VALUES (?)", [(text,), (text,)]) - expected = [text, text] - else: - cursor.execute("INSERT INTO #encoding_warning VALUES (?)", text) - expected = [text] - cursor.execute("SELECT data FROM #encoding_warning") - assert [row[0] for row in cursor.fetchall()] == expected + with ( + patch.object(ddbc, "DDBCSQLExecute", wraps=ddbc.DDBCSQLExecute) as execute, + patch.object(ddbc, "SQLExecuteMany", wraps=ddbc.SQLExecuteMany) as executemany, + ): + if method == "executemany": + cursor.executemany( + "INSERT INTO #encoding_warning VALUES (?)", [(text,), (text,)] + ) + expected = [text, text] + # Streaming batches must use execute()'s UTF-16 DAE path. + assert execute.call_count == (2 if large else 0) + assert executemany.call_count == (0 if large else 1) + else: + cursor.execute("INSERT INTO #encoding_warning VALUES (?)", text) + expected = [text] + assert execute.call_count == 1 + executemany.assert_not_called() + cursor.execute("SELECT data, CONVERT(VARBINARY(MAX), data) FROM #encoding_warning") + rows = cursor.fetchall() + assert [row[0] for row in rows] == expected + assert [row[1] for row in rows] == [value.encode("utf-16le") for value in expected] def test_dae_sql_c_char_with_various_data_types(db_connection): @@ -6022,33 +6058,6 @@ def test_dae_sql_c_char_with_various_data_types(db_connection): cursor.close() -def test_dae_encoding_error_handling(db_connection): - """Test DAE encoding error handling (lines 1751-1755).""" - db_connection.setencoding(encoding="ascii", ctype=mssql_python.SQL_CHAR) - - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_dae_error (id INT, data VARCHAR(MAX))") - - # Large non-ASCII string to trigger both DAE and encoding error - large_unicode = "你好" * 5000 - - error_raised = False - try: - cursor.execute("INSERT INTO #test_dae_error (id, data) VALUES (?, ?)", 1, large_unicode) - except (UnicodeEncodeError, RuntimeError, Exception) as e: - error_raised = True - error_msg = str(e).lower() - assert any(word in error_msg for word in ["encode", "ascii", "failed"]) - - # Should raise error in strict mode - if not error_raised: - pass # Some implementations may handle differently - - finally: - cursor.close() - - def test_executemany_sql_c_char_encoding_paths(db_connection): """Test executemany with SQL_C_CHAR encoding (lines 2043-2060).""" db_connection.setencoding(encoding="utf-8", ctype=mssql_python.SQL_CHAR) @@ -6416,25 +6425,6 @@ def test_cpp_dae_bytes_encoding(db_connection): cursor.close() -def test_cpp_dae_encoding_error(db_connection): - """encoding error in Data-At-Execution.""" - db_connection.setencoding(encoding="ascii", ctype=mssql_python.SQL_CHAR) - cursor = db_connection.cursor() - try: - cursor.execute("CREATE TABLE #test_cpp_dae_err (data VARCHAR(MAX))") - # Large non-ASCII string to trigger DAE + encoding error - large_unicode = "你好世界 " * 3000 - try: - cursor.execute("INSERT INTO #test_cpp_dae_err VALUES (?)", large_unicode) - # No error is OK - some implementations may handle it - except Exception as e: - # Expected: catch block lines 1753-1756 - error_msg = str(e).lower() - assert "encode" in error_msg or "ascii" in error_msg - finally: - cursor.close() - - def test_cpp_executemany_str_encoding(db_connection): """str encoding in executemany.""" db_connection.setencoding(encoding="utf-8", ctype=mssql_python.SQL_CHAR)