Fix seg fault using Python 2 invalid utf-8 strings and wstring
Fixes seg fault when passing a Python string, containing invalid utf-8 content, to a wstring or wchar * parameter. A TypeError is thrown instead, eg: %include <std_wstring.i> void instring(const std::wstring& s); instring(b"h\xe9llooo") # Python
This commit is contained in:
parent
3915e7bd08
commit
e96316bf31
5 changed files with 57 additions and 4 deletions
|
|
@ -7,6 +7,15 @@ the issue number to the end of the URL: https://github.com/swig/swig/issues/
|
|||
Version 4.0.0 (in progress)
|
||||
===========================
|
||||
|
||||
2018-06-15: wsfulton
|
||||
[Python] Fix seg fault using Python 2 when passing a Python string, containing
|
||||
invalid utf-8 content, to a wstring or wchar * parameter. A TypeError is thrown instead, eg:
|
||||
|
||||
%include <std_wstring.i>
|
||||
void instring(const std::wstring& s);
|
||||
|
||||
instring(b"h\xe9llooo") # Python
|
||||
|
||||
2018-06-12: wsfulton
|
||||
[Python] The %pythonabc feature in pyabc.i now uses base classes
|
||||
collections.abc.MutableSequence
|
||||
|
|
|
|||
|
|
@ -6501,6 +6501,7 @@ void instring(const char *s) {
|
|||
</pre></div>
|
||||
|
||||
<p>
|
||||
Note that "\xe9" is an invalid UTF-8 encoding, but "\xc3\xb6" is valid.
|
||||
When this method is called from Python 3, the return value is the following
|
||||
text string:
|
||||
</p>
|
||||
|
|
|
|||
|
|
@ -94,6 +94,14 @@ void test_throw() TESTCASE_THROW1(std::wstring){
|
|||
throw x;
|
||||
}
|
||||
|
||||
const char * non_utf8_c_str() {
|
||||
return "h\xe9llo";
|
||||
}
|
||||
|
||||
size_t size_wstring(const std::wstring& s) {
|
||||
return s.size();
|
||||
}
|
||||
|
||||
#ifdef SWIGPYTHON_BUILTIN
|
||||
bool is_python_builtin() { return true; }
|
||||
#else
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import li_std_wstring
|
||||
import sys
|
||||
|
||||
x = u"h"
|
||||
|
||||
|
|
@ -81,3 +82,25 @@ if b.name != "hello":
|
|||
b.a = li_std_wstring.A("hello")
|
||||
if b.a != u"hello":
|
||||
raise RuntimeError("bad string mapping")
|
||||
|
||||
# Byte strings only converted in Python 2
|
||||
if sys.version_info[0:2] < (3, 0):
|
||||
x = b"hello there"
|
||||
if li_std_wstring.test_value(x) != x:
|
||||
raise RuntimeError("bad string mapping")
|
||||
|
||||
# Invalid utf-8 in a byte string fails in all versions
|
||||
x = b"h\xe9llo"
|
||||
try:
|
||||
li_std_wstring.test_value(x)
|
||||
raise RuntimeError("TypeError not thrown")
|
||||
except TypeError:
|
||||
pass
|
||||
|
||||
# Check surrogateescape
|
||||
if sys.version_info[0:2] > (3, 1):
|
||||
x = u"h\udce9llo" # surrogate escaped representation of C char*: "h\xe9llo"
|
||||
if li_std_wstring.non_utf8_c_str() != x:
|
||||
raise RuntimeError("surrogateescape not working")
|
||||
if li_std_wstring.size_wstring(x) != 5 and len(x) != 5:
|
||||
raise RuntimeError("Unexpected length")
|
||||
|
|
|
|||
|
|
@ -18,16 +18,28 @@ SWIG_AsWCharPtrAndSize(PyObject *obj, wchar_t **cptr, size_t *psize, int *alloc)
|
|||
int isunicode = PyUnicode_Check(obj);
|
||||
%#if PY_VERSION_HEX < 0x03000000 && !defined(SWIG_PYTHON_STRICT_UNICODE_WCHAR)
|
||||
if (!isunicode && PyString_Check(obj)) {
|
||||
obj = tmp = PyUnicode_FromObject(obj);
|
||||
isunicode = 1;
|
||||
tmp = PyUnicode_FromObject(obj);
|
||||
if (tmp) {
|
||||
isunicode = 1;
|
||||
obj = tmp;
|
||||
} else {
|
||||
PyErr_Clear();
|
||||
return SWIG_TypeError;
|
||||
}
|
||||
}
|
||||
%#endif
|
||||
if (isunicode) {
|
||||
Py_ssize_t len = PyUnicode_GetSize(obj);
|
||||
if (cptr) {
|
||||
Py_ssize_t length;
|
||||
*cptr = %new_array(len + 1, wchar_t);
|
||||
PyUnicode_AsWideChar(SWIGPY_UNICODE_ARG(obj), *cptr, len);
|
||||
(*cptr)[len] = 0;
|
||||
length = PyUnicode_AsWideChar(SWIGPY_UNICODE_ARG(obj), *cptr, len);
|
||||
if (length == -1) {
|
||||
PyErr_Clear();
|
||||
Py_XDECREF(tmp);
|
||||
return SWIG_TypeError;
|
||||
}
|
||||
(*cptr)[length] = 0;
|
||||
}
|
||||
if (psize) *psize = (size_t) len + 1;
|
||||
if (alloc) *alloc = cptr ? SWIG_NEWOBJ : 0;
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue