diff --git a/Misc/NEWS.d/next/Library/2026-08-13-09-00-00.gh-issue-156241.eScFP1.rst b/Misc/NEWS.d/next/Library/2026-08-13-09-00-00.gh-issue-156241.eScFP1.rst new file mode 100644 index 00000000000000..dfe82854fa24ed --- /dev/null +++ b/Misc/NEWS.d/next/Library/2026-08-13-09-00-00.gh-issue-156241.eScFP1.rst @@ -0,0 +1 @@ +Speed up :func:`json.dump` for strings that need no escaping. diff --git a/Modules/_json.c b/Modules/_json.c index 3a724a3e72b185..2fbc75e8eeb8a1 100644 --- a/Modules/_json.c +++ b/Modules/_json.c @@ -223,6 +223,28 @@ ascii_escape_unicode_and_size(const void *input, int kind, Py_ssize_t input_char return rval; } +static PyObject * +quote_unescaped_unicode(PyObject *pystr) +{ + Py_ssize_t len = PyUnicode_GET_LENGTH(pystr); + PyObject *rval = PyUnicode_New(len + 2, PyUnicode_MAX_CHAR_VALUE(pystr)); + if (rval == NULL) { + return NULL; + } + int kind = PyUnicode_KIND(rval); + void *data = PyUnicode_DATA(rval); + /* No escapes means the output has the same maxchar class as the input, + so PyUnicode_New() gave us the same kind and a raw copy is valid. */ + assert(kind == PyUnicode_KIND(pystr)); + PyUnicode_WRITE(kind, data, 0, '"'); + memcpy((char *)data + kind, PyUnicode_DATA(pystr), (size_t)len * kind); + PyUnicode_WRITE(kind, data, len + 1, '"'); +#ifdef Py_DEBUG + assert(_PyUnicode_CheckConsistency(rval, 1)); +#endif + return rval; +} + static PyObject * ascii_escape_unicode(PyObject *pystr) { @@ -236,6 +258,10 @@ ascii_escape_unicode(PyObject *pystr) return NULL; } + if (output_size == input_chars + 2) { + return quote_unescaped_unicode(pystr); + } + return ascii_escape_unicode_and_size(input, kind, input_chars, output_size); } @@ -383,6 +409,10 @@ escape_unicode(PyObject *pystr) return NULL; } + if (output_size == input_chars + 2) { + return quote_unescaped_unicode(pystr); + } + return escape_unicode_and_size(input, kind, maxchar, input_chars, output_size); }