diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 09fe77a..079ed03 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -28,7 +28,7 @@ jobs: fail-fast: false matrix: os: [ubuntu-latest, windows-latest] - python_version: ["3.9", "3.10", "3.11", "3.12"] + python_version: ["3.9", "3.10", "3.11", "3.12", "3.13", "3.14"] runs-on: ${{ matrix.os }} steps: diff --git a/srsly/tests/ujson/test_ujson.py b/srsly/tests/ujson/test_ujson.py index fac9a78..f1af468 100644 --- a/srsly/tests/ujson/test_ujson.py +++ b/srsly/tests/ujson/test_ujson.py @@ -815,6 +815,41 @@ def test_sortKeys(self): sortedKeys = ujson.dumps(data, sort_keys=True) self.assertEqual(sortedKeys, '{"a":1,"b":1,"c":1,"d":1,"e":1,"f":1}') + def test_sortKeys_nested(self): + self.assertEqual(ujson.dumps({}, sort_keys=True), "{}") + data = {"b": {"d": 1, "c": [{"z": 0, "y": 1}, {}]}, "a": 2} + self.assertEqual( + ujson.dumps(data, sort_keys=True), + '{"a":2,"b":{"c":[{"y":1,"z":0},{}],"d":1}}', + ) + + def test_sortKeys_unorderable(self): + # Used to segfault instead of raising. + for data in ( + {1: 1, "a": 2}, + {"x": {1: 1, "a": 2}, "y": 3}, + [{"a": 1}, {1: 1, "a": 2}], + ): + with self.assertRaises(TypeError): + ujson.dumps(data, sort_keys=True) + + @unittest.skipIf(not hasattr(sys, 'getrefcount') == True, reason="test requires sys.refcount") + def test_sortKeys_does_not_leak(self): + import gc + + gc.collect() + key = "key" + str(id(self)) + value = ["abc"] + data = {key: value, "z": 1} + key_refs, value_refs = sys.getrefcount(key), sys.getrefcount(value) + for _ in range(100): + ujson.dumps(data, sort_keys=True) + with self.assertRaises(TypeError): + ujson.dumps({key: value, 1: 1}, sort_keys=True) + gc.collect() + self.assertEqual(key_refs, sys.getrefcount(key)) + self.assertEqual(value_refs, sys.getrefcount(value)) + @unittest.skipIf(not hasattr(sys, 'getrefcount') == True, reason="test requires sys.refcount") def test_does_not_leak_dictionary_values(self): import gc @@ -989,3 +1024,33 @@ def test_decode_surrogate_characters(test_input, expected): # Ensure that this matches stdlib's behaviour assert json.loads(test_input) == expected + + +@pytest.mark.parametrize( + "bad_value,error", + [ + (float("nan"), OverflowError), # error raised by the encoder + (object(), TypeError), # error raised by Python + ({1: 1, "a": 2}, TypeError), # unorderable keys with sort_keys + ], +) +def test_encode_error_does_not_leak_buffer(bad_value, error): + # The value is large enough that the output buffer is heap-allocated, + # and it was never freed when encoding then failed. + import tracemalloc + + data = ["x" * 100000, bad_value] + tracemalloc.start() + try: + for _ in range(5): + with pytest.raises(error): + ujson.dumps(data, sort_keys=True) + before = tracemalloc.get_traced_memory()[0] + for _ in range(100): + with pytest.raises(error): + ujson.dumps(data, sort_keys=True) + growth = tracemalloc.get_traced_memory()[0] - before + finally: + tracemalloc.stop() + # A leak is at least 100 * 100KB = 10MB. + assert growth < 1000000, growth diff --git a/srsly/ujson/lib/ultrajsonenc.c b/srsly/ujson/lib/ultrajsonenc.c index ea6372b..df4ec39 100644 --- a/srsly/ujson/lib/ultrajsonenc.c +++ b/srsly/ujson/lib/ultrajsonenc.c @@ -755,7 +755,7 @@ void encode(JSOBJ obj, JSONObjectEncoder *enc, const char *name, size_t cbName) { const char *value; char *objName; - int count; + int count, res; JSOBJ iterObj; size_t szlen; JSONTypeContext tc; @@ -860,8 +860,19 @@ void encode(JSOBJ obj, JSONObjectEncoder *enc, const char *name, size_t cbName) Buffer_AppendCharUnchecked (enc, '{'); Buffer_AppendIndentNewlineUnchecked (enc); - while (enc->iterNext(obj, &tc)) + while ((res = enc->iterNext(obj, &tc))) { + if (res < 0) + { + // iterNext failed (e.g. sort_keys with unorderable keys). Stop + // encoding so nothing else runs with the Python error pending. + SetError (obj, enc, "Failed to iterate over object"); + enc->iterEnd(obj, &tc); + enc->endTypeContext(obj, &tc); + enc->level --; + return; + } + // The extra 2 bytes cover the comma and optional newline. Buffer_Reserve (enc, enc->indent * (enc->level + 1) + 2); @@ -1059,6 +1070,11 @@ char *JSON_EncodeObject(JSOBJ obj, JSONObjectEncoder *enc, char *_buffer, size_t Buffer_Reserve(enc, 1); if (enc->errorMsg) { + if (enc->heap == 1) + { + // Buffer was realloc'd at some point, or no initial buffer was provided. + enc->free(enc->start); + } return NULL; } Buffer_AppendCharUnchecked(enc, '\0'); diff --git a/srsly/ujson/objToJSON.c b/srsly/ujson/objToJSON.c index c22f861..8d47eb0 100644 --- a/srsly/ujson/objToJSON.c +++ b/srsly/ujson/objToJSON.c @@ -394,11 +394,22 @@ int SortedDict_iterNext(JSOBJ obj, JSONTypeContext *tc) } // Obtain the value for each key, and pack a list of (key, value) 2-tuples. + // On every path to error, key and item have already been released or + // handed over, so only items needs cleaning up. nitems = PyList_GET_SIZE(items); for (i = 0; i < nitems; i++) { key = PyList_GET_ITEM(items, i); + // Borrowed reference. value = PyDict_GetItem(GET_TC(tc)->dictObj, key); + if (value == NULL) + { + if (!PyErr_Occurred()) + { + PyErr_SetObject(PyExc_KeyError, key); + } + goto error; + } // Subject the key to the same type restrictions and conversions as in Dict_iterGetValue. if (PyUnicode_Check(key)) @@ -409,26 +420,34 @@ int SortedDict_iterNext(JSOBJ obj, JSONTypeContext *tc) { key = PyObject_Str(key); #if PY_MAJOR_VERSION >= 3 - keyTmp = key; - key = PyUnicode_AsUTF8String(key); - Py_DECREF(keyTmp); + if (key != NULL) + { + keyTmp = key; + key = PyUnicode_AsUTF8String(key); + Py_DECREF(keyTmp); + } #endif } else { Py_INCREF(key); } + if (key == NULL) + { + goto error; + } item = PyTuple_Pack(2, key, value); + Py_DECREF(key); if (item == NULL) { goto error; } + // Steals the reference to item, even on failure. if (PyList_SetItem(items, i, item)) { goto error; } - Py_DECREF(key); } // Store the sorted list of tuples in the newObj slot. @@ -449,9 +468,6 @@ int SortedDict_iterNext(JSOBJ obj, JSONTypeContext *tc) return 1; error: - Py_XDECREF(item); - Py_XDECREF(key); - Py_XDECREF(value); Py_XDECREF(items); return -1; } @@ -460,7 +476,7 @@ void SortedDict_iterEnd(JSOBJ obj, JSONTypeContext *tc) { GET_TC(tc)->itemName = NULL; GET_TC(tc)->itemValue = NULL; - Py_DECREF(GET_TC(tc)->newObj); + // newObj (the sorted items list) is released in Object_endTypeContext. Py_DECREF(GET_TC(tc)->dictObj); PRINTMARK(); } @@ -887,19 +903,19 @@ PyObject* objToJSON(PyObject* self, PyObject *args, PyObject *kwargs) ret = JSON_EncodeObject (oinput, &encoder, buffer, sizeof (buffer)); PRINTMARK(); - if (PyErr_Occurred()) + if (encoder.errorMsg && !PyErr_Occurred()) { - return NULL; + // If there is an error message and we don't already have a Python exception, set one. + PyErr_Format (PyExc_OverflowError, "%s", encoder.errorMsg); } - if (encoder.errorMsg) + if (PyErr_Occurred()) { if (ret != buffer) { encoder.free (ret); } - PyErr_Format (PyExc_OverflowError, "%s", encoder.errorMsg); return NULL; }