codekingpro/portable-devtools
114k
1/* C implementation of performance sensitive functions. */
2
3#define PY_SSIZE_T_CLEAN
4#include <Python.h>
5#include <stdint.h> /* uint8_t, uint32_t, uint64_t */
6
7#if __ARM_NEON
8#include <arm_neon.h>
9#elif __SSE2__
10#include <emmintrin.h>
11#endif
12
13static const Py_ssize_t MASK_LEN = 4;
14
15/* Similar to PyBytes_AsStringAndSize, but accepts more types */
16
17static int
18_PyBytesLike_AsStringAndSize(PyObject *obj, PyObject **tmp, char **buffer, Py_ssize_t *length)
19{
20 // This supports bytes, bytearrays, and memoryview objects,
21 // which are common data structures for handling byte streams.
22 // If *tmp isn't NULL, the caller gets a new reference.
23 if (PyBytes_Check(obj))
24 {
25 *tmp = NULL;
26 *buffer = PyBytes_AS_STRING(obj);
27 *length = PyBytes_GET_SIZE(obj);
28 }
29 else if (PyByteArray_Check(obj))
30 {
31 *tmp = NULL;
32 *buffer = PyByteArray_AS_STRING(obj);
33 *length = PyByteArray_GET_SIZE(obj);
34 }
35 else if (PyMemoryView_Check(obj))
36 {
37 *tmp = PyMemoryView_GetContiguous(obj, PyBUF_READ, 'C');
38 if (*tmp == NULL)
39 {
40 return -1;
41 }
42 Py_buffer *mv_buf;
43 mv_buf = PyMemoryView_GET_BUFFER(*tmp);
44 *buffer = mv_buf->buf;
45 *length = mv_buf->len;
46 }
47 else
48 {
49 PyErr_Format(
50 PyExc_TypeError,
51 "expected a bytes-like object, %.200s found",
52 Py_TYPE(obj)->tp_name);
53 return -1;
54 }
55
56 return 0;
57}
58
59/* C implementation of websockets.utils.apply_mask */
60
61static PyObject *
62apply_mask(PyObject *self, PyObject *args, PyObject *kwds)
63{
64
65 // In order to support various bytes-like types, accept any Python object.
66
67 static char *kwlist[] = {"data", "mask", NULL};
68 PyObject *input_obj;
69 PyObject *mask_obj;
70
71 // A pointer to a char * + length will be extracted from the data and mask
72 // arguments, possibly via a Py_buffer.
73
74 PyObject *input_tmp = NULL;
75 char *input;
76 Py_ssize_t input_len;
77 PyObject *mask_tmp = NULL;
78 char *mask;
79 Py_ssize_t mask_len;
80
81 // Initialize a PyBytesObject then get a pointer to the underlying char *
82 // in order to avoid an extra memory copy in PyBytes_FromStringAndSize.
83
84 PyObject *result = NULL;
85 char *output;
86
87 // Other variables.
88
89 Py_ssize_t i = 0;
90
91 // Parse inputs.
92
93 if (!PyArg_ParseTupleAndKeywords(
94 args, kwds, "OO", kwlist, &input_obj, &mask_obj))
95 {
96 goto exit;
97 }
98
99 if (_PyBytesLike_AsStringAndSize(input_obj, &input_tmp, &input, &input_len) == -1)
100 {
101 goto exit;
102 }
103
104 if (_PyBytesLike_AsStringAndSize(mask_obj, &mask_tmp, &mask, &mask_len) == -1)
105 {
106 goto exit;
107 }
108
109 if (mask_len != MASK_LEN)
110 {
111 PyErr_SetString(PyExc_ValueError, "mask must contain 4 bytes");
112 goto exit;
113 }
114
115 // Create output.
116
117 result = PyBytes_FromStringAndSize(NULL, input_len);
118 if (result == NULL)
119 {
120 goto exit;
121 }
122
123 // Since we just created result, we don't need error checks.
124 output = PyBytes_AS_STRING(result);
125
126 // Perform the masking operation.
127
128 // Apparently GCC cannot figure out the following optimizations by itself.
129
130 // We need a new scope for MSVC 2010 (non C99 friendly)
131 {
132#if __ARM_NEON
133
134 // With NEON support, XOR by blocks of 16 bytes = 128 bits.
135
136 Py_ssize_t input_len_128 = input_len & ~15;
137 uint8x16_t mask_128 = vreinterpretq_u8_u32(vdupq_n_u32(*(uint32_t *)mask));
138
139 for (; i < input_len_128; i += 16)
140 {
141 uint8x16_t in_128 = vld1q_u8((uint8_t *)(input + i));
142 uint8x16_t out_128 = veorq_u8(in_128, mask_128);
143 vst1q_u8((uint8_t *)(output + i), out_128);
144 }
145
146#elif __SSE2__
147
148 // With SSE2 support, XOR by blocks of 16 bytes = 128 bits.
149
150 // Since we cannot control the 16-bytes alignment of input and output
151 // buffers, we rely on loadu/storeu rather than load/store.
152
153 Py_ssize_t input_len_128 = input_len & ~15;
154 __m128i mask_128 = _mm_set1_epi32(*(uint32_t *)mask);
155
156 for (; i < input_len_128; i += 16)
157 {
158 __m128i in_128 = _mm_loadu_si128((__m128i *)(input + i));
159 __m128i out_128 = _mm_xor_si128(in_128, mask_128);
160 _mm_storeu_si128((__m128i *)(output + i), out_128);
161 }
162
163#else
164
165 // Without SSE2 support, XOR by blocks of 8 bytes = 64 bits.
166
167 // We assume the memory allocator aligns everything on 8 bytes boundaries.
168
169 Py_ssize_t input_len_64 = input_len & ~7;
170 uint32_t mask_32 = *(uint32_t *)mask;
171 uint64_t mask_64 = ((uint64_t)mask_32 << 32) | (uint64_t)mask_32;
172
173 for (; i < input_len_64; i += 8)
174 {
175 *(uint64_t *)(output + i) = *(uint64_t *)(input + i) ^ mask_64;
176 }
177
178#endif
179 }
180
181 // XOR the remainder of the input byte by byte.
182
183 for (; i < input_len; i++)
184 {
185 output[i] = input[i] ^ mask[i & (MASK_LEN - 1)];
186 }
187
188exit:
189 Py_XDECREF(input_tmp);
190 Py_XDECREF(mask_tmp);
191 return result;
192
193}
194
195static PyMethodDef speedups_methods[] = {
196 {
197 "apply_mask",
198 (PyCFunction)apply_mask,
199 METH_VARARGS | METH_KEYWORDS,
200 "Apply masking to the data of a WebSocket message.",
201 },
202 {NULL, NULL, 0, NULL}, /* Sentinel */
203};
204
205static struct PyModuleDef speedups_module = {
206 PyModuleDef_HEAD_INIT,
207 "websocket.speedups", /* m_name */
208 "C implementation of performance sensitive functions.",
209 /* m_doc */
210 -1, /* m_size */
211 speedups_methods, /* m_methods */
212 NULL,
213 NULL,
214 NULL,
215 NULL
216};
217
218PyMODINIT_FUNC
219PyInit_speedups(void)
220{
221 PyObject *m = PyModule_Create(&speedups_module);
222 if (m == NULL) {
223 return NULL;
224 }
225#ifdef Py_GIL_DISABLED
226 PyUnstable_Module_SetGIL(m, Py_MOD_GIL_NOT_USED);
227#endif
228 return m;
229}
230 