Fix seg fault using Python 2 invalid utf-8 strings and wstring

Fixes seg fault when passing a Python string, containing invalid utf-8 content,
to a wstring or wchar * parameter.  A TypeError is thrown instead, eg:

  %include <std_wstring.i>
  void instring(const std::wstring& s);

  instring(b"h\xe9llooo") # Python
This commit is contained in:
William S Fulton 2018-06-15 19:14:52 +01:00
commit e96316bf31
5 changed files with 57 additions and 4 deletions

View file

@ -7,6 +7,15 @@ the issue number to the end of the URL: https://github.com/swig/swig/issues/
Version 4.0.0 (in progress)
===========================
2018-06-15: wsfulton
[Python] Fix seg fault using Python 2 when passing a Python string, containing
invalid utf-8 content, to a wstring or wchar * parameter. A TypeError is thrown instead, eg:
%include <std_wstring.i>
void instring(const std::wstring& s);
instring(b"h\xe9llooo") # Python
2018-06-12: wsfulton
[Python] The %pythonabc feature in pyabc.i now uses base classes
collections.abc.MutableSequence

View file

@ -6501,6 +6501,7 @@ void instring(const char *s) {
</pre></div>
<p>
Note that "\xe9" is an invalid UTF-8 encoding, but "\xc3\xb6" is valid.
When this method is called from Python 3, the return value is the following
text string:
</p>

View file

@ -94,6 +94,14 @@ void test_throw() TESTCASE_THROW1(std::wstring){
throw x;
}
const char * non_utf8_c_str() {
return "h\xe9llo";
}
size_t size_wstring(const std::wstring& s) {
return s.size();
}
#ifdef SWIGPYTHON_BUILTIN
bool is_python_builtin() { return true; }
#else

View file

@ -1,4 +1,5 @@
import li_std_wstring
import sys
x = u"h"
@ -81,3 +82,25 @@ if b.name != "hello":
b.a = li_std_wstring.A("hello")
if b.a != u"hello":
raise RuntimeError("bad string mapping")
# Byte strings only converted in Python 2
if sys.version_info[0:2] < (3, 0):
x = b"hello there"
if li_std_wstring.test_value(x) != x:
raise RuntimeError("bad string mapping")
# Invalid utf-8 in a byte string fails in all versions
x = b"h\xe9llo"
try:
li_std_wstring.test_value(x)
raise RuntimeError("TypeError not thrown")
except TypeError:
pass
# Check surrogateescape
if sys.version_info[0:2] > (3, 1):
x = u"h\udce9llo" # surrogate escaped representation of C char*: "h\xe9llo"
if li_std_wstring.non_utf8_c_str() != x:
raise RuntimeError("surrogateescape not working")
if li_std_wstring.size_wstring(x) != 5 and len(x) != 5:
raise RuntimeError("Unexpected length")

View file

@ -18,16 +18,28 @@ SWIG_AsWCharPtrAndSize(PyObject *obj, wchar_t **cptr, size_t *psize, int *alloc)
int isunicode = PyUnicode_Check(obj);
%#if PY_VERSION_HEX < 0x03000000 && !defined(SWIG_PYTHON_STRICT_UNICODE_WCHAR)
if (!isunicode && PyString_Check(obj)) {
obj = tmp = PyUnicode_FromObject(obj);
isunicode = 1;
tmp = PyUnicode_FromObject(obj);
if (tmp) {
isunicode = 1;
obj = tmp;
} else {
PyErr_Clear();
return SWIG_TypeError;
}
}
%#endif
if (isunicode) {
Py_ssize_t len = PyUnicode_GetSize(obj);
if (cptr) {
Py_ssize_t length;
*cptr = %new_array(len + 1, wchar_t);
PyUnicode_AsWideChar(SWIGPY_UNICODE_ARG(obj), *cptr, len);
(*cptr)[len] = 0;
length = PyUnicode_AsWideChar(SWIGPY_UNICODE_ARG(obj), *cptr, len);
if (length == -1) {
PyErr_Clear();
Py_XDECREF(tmp);
return SWIG_TypeError;
}
(*cptr)[length] = 0;
}
if (psize) *psize = (size_t) len + 1;
if (alloc) *alloc = cptr ? SWIG_NEWOBJ : 0;