FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

GitHub Viewer

#include #include "pycore_bytesobject.h" // _PyBytes_DecodeEscape() #include "pycore_unicodeobject.h" // _PyUnicode_DecodeUnicodeEscapeInternal() #include "pegen.h" #include "string_parser.h" #include //// STRING HANDLING FUNCTIONS //// static int warn_invalid_escape_sequence(Parser *p, const char* buffer, const char *first_invalid_escape, Token *t) { if (p->call_invalid_rules) { // Do not report warnings if we are in the second pass of the parser // to avoid showing the warning twice. return 0; } unsigned char c = (unsigned char)*first_invalid_escape; if ((t->type == FSTRING_MIDDLE || t->type == FSTRING_END || t->type == TSTRING_MIDDLE || t->type == TSTRING_END) && (c == '{' || c == '}')) { // in this case the tokenizer has already emitted a warning, // see Parser/tokenizer/helpers.c:warn_invalid_escape_sequence return 0; } int octal = ('4' = 12) { category = PyExc_SyntaxWarning; } else { category = PyExc_DeprecationWarning; } // Calculate the lineno and the col_offset of the invalid escape sequence const char *start = buffer; const char *end = first_invalid_escape; int lineno = t->lineno; int col_offset = t->col_offset; while (start < end) { if (*start == '\n') { lineno++; col_offset = 0; } else { col_offset++; } start++; } // Count the number of quotes in the token char first_quote = 0; if (lineno == t->lineno) { int quote_count = 0; const char* tok = PyBytes_AsString(t->bytes); for (int i = 0; i < PyBytes_Size(t->bytes); i++) { if (tok[i] == '\'' || tok[i] == '\"') { if (quote_count == 0) { first_quote = tok[i]; } if (tok[i] == first_quote) { quote_count++; } } else { break; } } col_offset += quote_count; } _PyTokenizer_Info info = _PyTokenizer_GetInfo(p->tok); if (PyErr_WarnExplicitObject(category, msg, info.filename, lineno, info.module, NULL) < 0) { if (PyErr_ExceptionMatches(category)) { /* Replace the Syntax/DeprecationWarning exception with a SyntaxError to get a more accurate error report */ PyErr_Clear(); /* This is needed, in order for the SyntaxError to point to the token t, since _PyPegen_raise_error uses p->tokens[p->fill - 1] for the error location, if p->known_err_token is not set. */ p->known_err_token = t; if (octal) { RAISE_ERROR_KNOWN_LOCATION(p, PyExc_SyntaxError, lineno, col_offset-1, lineno, col_offset+1, "\"\\%.3s\" is an invalid octal escape sequence. " "Did you mean \"\\\\%.3s\"? A raw string is also an option.", first_invalid_escape, first_invalid_escape); } else { RAISE_ERROR_KNOWN_LOCATION(p, PyExc_SyntaxError, lineno, col_offset-1, lineno, col_offset+1, "\"\\%c\" is an invalid escape sequence. " "Did you mean \"\\\\%c\"? A raw string is also an option.", c, c); } } Py_DECREF(msg); return -1; } Py_DECREF(msg); return 0; } static PyObject * decode_utf8(const char **sPtr, const char *end) { const char *s; const char *t; t = s = *sPtr; while (s < end && (*s & 0x80)) { s++; } *sPtr = s; return PyUnicode_DecodeUTF8(t, s - t, NULL); } static PyObject * decode_unicode_with_escapes(Parser *parser, const char *s, size_t len, Token *t) { PyObject *v; char *p; const char *end; /* check for integer overflow */ if (len > (size_t)PY_SSIZE_T_MAX / 6) { return NULL; } /* "ä" (2 bytes) may become "\U000000E4" (10 bytes), or 1:5. * "\ä" (3 bytes) may become "\u005c\U000000E4" (16 bytes), or ~1:6. */ Py_ssize_t alloc = (Py_ssize_t)len * 6; char *buf = PyMem_Malloc(alloc); if (buf == NULL) { return PyErr_NoMemory(); } p = buf; end = s + len; while (s < end) { if (*s == '\\') { *p++ = *s++; if (s >= end || *s & 0x80) { memcpy(p, "u005c", 5); p += 5; if (s >= end) { break; } } } if (*s & 0x80) { PyObject *w; int kind; const void *data; Py_ssize_t w_len; Py_ssize_t i; w = decode_utf8(&s, end); if (w == NULL) { PyMem_Free(buf); return NULL; } kind = PyUnicode_KIND(w); data = PyUnicode_DATA(w); w_len = PyUnicode_GET_LENGTH(w); for (i = 0; i < w_len; i++) { // sprintf() writes a null byte: the buffer is large enough // for that thanks to the overallocation. assert((p + 11 - buf) = 1); if (len > INT_MAX) { PyErr_SetString(PyExc_OverflowError, "string to parse is too long"); return NULL; } if (s[--len] != quote) { /* Last quote char must match the first. */ PyErr_BadInternalCall(); return NULL; } if (len >= 4 && s[0] == quote && s[1] == quote) { /* A triple quoted string. We've already skipped one quote at the start and one at the end of the string. Now skip the two at the start. */ s += 2; len -= 2; /* And check that the last two match. */ if (s[--len] != quote || s[--len] != quote) { PyErr_BadInternalCall(); return NULL; } } /* Avoid invoking escape decoding routines if possible. */ rawmode = rawmode || strchr(s, '\\') == NULL; if (bytesmode) { /* Disallow non-ASCII characters. */ const char *ch; for (ch = s; *ch; ch++) { if (Py_CHARMASK(*ch) >= 0x80) { RAISE_SYNTAX_ERROR_KNOWN_LOCATION( t, "bytes can only contain ASCII " "literal characters"); return NULL; } } if (rawmode) { return PyBytes_FromStringAndSize(s, (Py_ssize_t)len); } return decode_bytes_with_escapes(p, s, (Py_ssize_t)len, t); } return _PyPegen_decode_string(p, rawmode, s, len, t); }

Back | FazBrowse Home | New Git URL