semeru/code-text-galeras-commit-generation-3k-deduped
026
1[2 {3 "id": 125184,4 "commit_id": "adf24bfa9723b0621183bb27f0c889b813c06e8a",5 "repo": "ray",6 "path": "python/ray/_private/thirdparty/tabulate/tabulate.py",7 "file_name": "tabulate.py",8 "fun_name": "_format",9 "commit_message": "[State Observability] Use a table format by default (#26159)\n\nNOTE: tabulate is copied/pasted to the codebase for table formatting.\r\n\r\nThis PR changes the default layout to be the table format for both summary and list APIs.",10 "code": "def _format(val, valtype, floatfmt, missingval=\"\", has_invisible=True):\n # noqa\n if val is None:\n return missingval\n\n if valtype in [int, _text_type]:\n return \"{0}\".format(val)\n elif valtype is _binary_type:\n try:\n return _text_type(val, \"ascii\")\n except TypeError:\n return _text_type(val)\n elif valtype is float:\n is_a_colored_number = has_invisible and isinstance(\n val, (_text_type, _binary_type)\n )\n if is_a_colored_number:\n raw_val = _strip_invisible(val)\n formatted_val = format(float(raw_val), floatfmt)\n return val.replace(raw_val, formatted_val)\n else:\n return format(float(val), floatfmt)\n else:\n return \"{0}\".format(val)\n\n",11 "url": "https://github.com/ray-project/ray.git",12 "language": "Python",13 "ast_errors": "",14 "n_ast_errors": 0,15 "ast_levels": 15,16 "n_whitespaces": 224,17 "n_words": 65,18 "vocab_size": 47,19 "complexity": 8,20 "nloc": 22,21 "token_counts": 132,22 "n_ast_nodes": 251,23 "n_identifiers": 1824 },25 {26 "id": 42070,27 "commit_id": "34662f4be5c364e7518f9c1118c9b362038ee5dd",28 "repo": "seaborn",29 "path": "seaborn/rcmod.py",30 "file_name": "rcmod.py",31 "fun_name": "set_context",32 "commit_message": "Convert docs to pydata-sphinx-theme and add new material (#2842)\n\n* Do basic conversion of site to pydata_sphinx_theme\r\n\r\n* Remove some pae structure customizations we no longer need\r\n\r\n* Add some custom CSS\r\n\r\n* Tweak a few more colors\r\n\r\n* Remove vestigial div closing tag\r\n\r\n* Reorganize release notes into hierarchical pages\r\n\r\n* Rebuild full docs and fix some resulting issues\r\n\r\n* Make release note doc refs absolute\r\n\r\n* Convert homepage to use sphinx-design instead of hand-crafted html\r\n\r\n* Remove original custom css\r\n\r\n* Simplify header and put archive switcher in footer\r\n\r\n* Streamline API docs for objects\r\n\r\n* Play around with templates to fix shrinking content (not perfect yet)\r\n\r\n* Improve use of horizontal space without sidebars\r\n\r\n* Various tweaks\r\n\r\n* Convert tutorial homepage source to native sphinx-design directives\r\n\r\n* Move intro page into tutorial\r\n\r\n* More tweaks\r\n\r\n* Tweak theme colors and footer\r\n\r\n* Remove reference to navbar version\r\n\r\n* Note that error bar tutorial demonstrates new features as of v0.12\r\n\r\n* Update layout customization for new theme features\r\n\r\n* Various layout and CSS tweaks\r\n\r\n* Narrow support guidance to StackOverflow\r\n\r\n* Run all notebooks\r\n\r\n* Adapt to new dropdown navbar in pydata theme\r\n\r\n* Separate tutorial source and outputs\r\n\r\n* Separate dostring source and outputs\r\n\r\n* Add scale API template\r\n\r\n* Update API docs\r\n\r\n* Fix requirements\r\n\r\n* Add new objects\r\n\r\n* Point doc requirements at v0.10 RC for theme",33 "code": "def set_context(context=None, font_scale=1, rc=None):\n \n context_object = plotting_context(context, font_scale, rc)\n mpl.rcParams.update(context_object)\n\n",34 "url": "https://github.com/mwaskom/seaborn.git",35 "language": "Python",36 "ast_errors": "",37 "n_ast_errors": 0,38 "ast_levels": 8,39 "n_whitespaces": 19,40 "n_words": 10,41 "vocab_size": 10,42 "complexity": 1,43 "nloc": 3,44 "token_counts": 34,45 "n_ast_nodes": 53,46 "n_identifiers": 947 },48 {49 "id": 311030,50 "commit_id": "5d7d652237b2368320a68c772ce3d837e4c1d04b",51 "repo": "core",52 "path": "homeassistant/components/synology_dsm/common.py",53 "file_name": "common.py",54 "fun_name": "async_unload",55 "commit_message": "Replace Synology DSM services with buttons (#57352)",56 "code": "async def async_unload(self) -> None:\n \n await self._syno_api_executer(self.dsm.logout)\n",57 "url": "https://github.com/home-assistant/core.git",58 "language": "Python",59 "ast_errors": "",60 "n_ast_errors": 0,61 "ast_levels": 10,62 "n_whitespaces": 21,63 "n_words": 7,64 "vocab_size": 7,65 "complexity": 1,66 "nloc": 3,67 "token_counts": 19,68 "n_ast_nodes": 35,69 "n_identifiers": 570 },71 {72 "id": 278720,73 "commit_id": "3613c3defc39c236fb1592c4f7ba1a9cc887343a",74 "repo": "keras",75 "path": "keras/engine/functional_utils.py",76 "file_name": "functional_utils.py",77 "fun_name": "clone_keras_tensors",78 "commit_message": "Remove pylint comments.\n\nPiperOrigin-RevId: 452353044",79 "code": "def clone_keras_tensors(args, keras_tensor_mapping):\n \n result = []\n for obj in tf.nest.flatten(args):\n if node_module.is_keras_tensor(obj):\n if id(obj) in keras_tensor_mapping:\n cpy = keras_tensor_mapping[id(obj)]\n else:\n # Create copy of keras_tensor if we haven't done it before\n cpy = _clone_keras_tensor(obj)\n cpy._keras_history = obj._keras_history\n keras_tensor_mapping[id(obj)] = cpy\n result.append(cpy)\n else:\n result.append(obj)\n return tf.nest.pack_sequence_as(args, result)\n\n",80 "url": "https://github.com/keras-team/keras.git",81 "language": "Python",82 "ast_errors": "",83 "n_ast_errors": 0,84 "ast_levels": 16,85 "n_whitespaces": 191,86 "n_words": 46,87 "vocab_size": 35,88 "complexity": 4,89 "nloc": 14,90 "token_counts": 98,91 "n_ast_nodes": 160,92 "n_identifiers": 1693 },94 {95 "id": 22168,96 "commit_id": "cd5a9683be69c86c8f3adcd13385a9bc5db198ec",97 "repo": "pipenv",98 "path": "pipenv/patched/pip/_vendor/rich/box.py",99 "file_name": "box.py",100 "fun_name": "get_plain_headed_box",101 "commit_message": "Rename notpip to pip. Vendor in pip-22.2.1 and latest requirementslib and vistir.",102 "code": "def get_plain_headed_box(self) -> \"Box\":\n \n return PLAIN_HEADED_SUBSTITUTIONS.get(self, self)\n",103 "url": "https://github.com/pypa/pipenv.git",104 "language": "Python",105 "ast_errors": "",106 "n_ast_errors": 0,107 "ast_levels": 7,108 "n_whitespaces": 21,109 "n_words": 7,110 "vocab_size": 7,111 "complexity": 1,112 "nloc": 9,113 "token_counts": 17,114 "n_ast_nodes": 31,115 "n_identifiers": 4116 },117 {118 "id": 156567,119 "commit_id": "2b90415b02d3ad1b08362889e0818590ca3133f4",120 "repo": "dask",121 "path": "dask/array/core.py",122 "file_name": "core.py",123 "fun_name": "apply_and_enforce",124 "commit_message": "Add kwarg ``enforce_ndim`` to ``dask.array.map_blocks()`` (#8865)",125 "code": "def apply_and_enforce(*args, **kwargs):\n \n func = kwargs.pop(\"_func\")\n expected_ndim = kwargs.pop(\"expected_ndim\")\n out = func(*args, **kwargs)\n if getattr(out, \"ndim\", 0) != expected_ndim:\n out_ndim = getattr(out, \"ndim\", 0)\n raise ValueError(\n f\"Dimension mismatch: expected output of {func} \"\n f\"to have dims = {expected_ndim}. Got {out_ndim} instead.\"\n )\n return out\n\n",126 "url": "https://github.com/dask/dask.git",127 "language": "Python",128 "ast_errors": "",129 "n_ast_errors": 0,130 "ast_levels": 12,131 "n_whitespaces": 106,132 "n_words": 44,133 "vocab_size": 36,134 "complexity": 2,135 "nloc": 11,136 "token_counts": 68,137 "n_ast_nodes": 129,138 "n_identifiers": 10139 },140 {141 "id": 178165,142 "commit_id": "283628097a10e8abafc94c683bc8be2d79a5998f",143 "repo": "label-studio",144 "path": "label_studio/core/redis.py",145 "file_name": "redis.py",146 "fun_name": "get_jobs_by_meta",147 "commit_message": "feat: DEV-2075: Add mixin to Project to support mechanism to cancel old jobs (#2547)\n\n* feat: DEV-2075: Add mixin to Project to support mechanism to cancel old jobs",148 "code": "def get_jobs_by_meta(queue, func_name, meta):\n \n # get all jobs from Queue\n jobs = (job\n for job in queue.get_jobs()\n if job.func.__name__ == func_name\n )\n # return only with same meta data\n return [job for job in jobs if hasattr(job, 'meta') and job.meta == meta]\n\n",149 "url": "https://github.com/heartexlabs/label-studio.git",150 "language": "Python",151 "ast_errors": "",152 "n_ast_errors": 0,153 "ast_levels": 11,154 "n_whitespaces": 90,155 "n_words": 42,156 "vocab_size": 33,157 "complexity": 6,158 "nloc": 6,159 "token_counts": 52,160 "n_ast_nodes": 83,161 "n_identifiers": 10162 },163 {164 "id": 269601,165 "commit_id": "84afc5193d38057e2e2badf9c889ea87d80d8fbf",166 "repo": "keras",167 "path": "keras/backend.py",168 "file_name": "backend.py",169 "fun_name": "enable_tf_random_generator",170 "commit_message": "Reformatting the codebase with black.\n\nPiperOrigin-RevId: 450093126",171 "code": "def enable_tf_random_generator():\n \n\n global _USE_GENERATOR_FOR_RNG\n _USE_GENERATOR_FOR_RNG = True\n\n\n@keras_export(\"keras.backend.experimental.disable_tf_random_generator\", v1=[])",172 "url": "https://github.com/keras-team/keras.git",173 "language": "Python",174 "ast_errors": "@keras_export(\"keras.backend.experimental.disable_tf_random_generator\", v1=[])",175 "n_ast_errors": 1,176 "ast_levels": 8,177 "n_whitespaces": 17,178 "n_words": 9,179 "vocab_size": 8,180 "complexity": 1,181 "nloc": 3,182 "token_counts": 10,183 "n_ast_nodes": 38,184 "n_identifiers": 4185 },186 {187 "id": 42548,188 "commit_id": "8a4cf5d94eb94b6427c5d1d7907ba07b119932c5",189 "repo": "nltk",190 "path": "nltk/text.py",191 "file_name": "text.py",192 "fun_name": "collocation_list",193 "commit_message": "Docstring tests (#3050)\n\n* fixed pytests\r\n\r\n* fixed more pytests\r\n\r\n* fixed more pytest and changed multiline pytest issues fixes for snowball.py and causal.py\r\n\r\n* fixed pytests (mainly multiline or rounding issues)\r\n\r\n* fixed treebank pytests, removed test for return_string=True (deprecated)\r\n\r\n* fixed destructive.py pytests, removed test for return_string=True (deprecated)\r\n\r\n* fixed pytest (rounding issues)\r\n\r\n* fixed pytest (initialised missing object)\r\n\r\n* fixed pytest (formatting issues)\r\n\r\n* fixed pytest (formatting issues)\r\n\r\n* fixed pytest (formatting issues)\r\n\r\n* added pytest +SKIP for deprecated module stanford\r\n\r\n* updated AUTHORS.md\r\n\r\n* changed docstring corrections by usage of ELLIPSIS and different roundings\r\n\r\n* fixed AUTHORS.md to be consistent\r\n\r\n* Fix framenet doctest formatting with pprint\r\n\r\n* Change docstring on MultiListBox.__init__\r\n\r\nI believe the original typo was misinterpreted and changed to something that was not originally intended.\r\n\r\nCo-authored-by: Jan Lennartz <jan.lennartz@ing.com>\r\nCo-authored-by: Tom Aarsen <37621491+tomaarsen@users.noreply.github.com>\r\nCo-authored-by: Tom Aarsen <Cubiegamedev@gmail.com>",194 "code": "def collocation_list(self, num=20, window_size=2):\n \n if not (\n \"_collocations\" in self.__dict__\n and self._num == num\n and self._window_size == window_size\n ):\n self._num = num\n self._window_size = window_size\n\n # print(\"Building collocations list\")\n from nltk.corpus import stopwords\n\n ignored_words = stopwords.words(\"english\")\n finder = BigramCollocationFinder.from_words(self.tokens, window_size)\n finder.apply_freq_filter(2)\n finder.apply_word_filter(lambda w: len(w) < 3 or w.lower() in ignored_words)\n bigram_measures = BigramAssocMeasures()\n self._collocations = list(\n finder.nbest(bigram_measures.likelihood_ratio, num)\n )\n return self._collocations\n",195 "url": "https://github.com/nltk/nltk.git",196 "language": "Python",197 "ast_errors": "",198 "n_ast_errors": 0,199 "ast_levels": 14,200 "n_whitespaces": 258,201 "n_words": 61,202 "vocab_size": 48,203 "complexity": 5,204 "nloc": 18,205 "token_counts": 126,206 "n_ast_nodes": 205,207 "n_identifiers": 27208 },209 {210 "id": 294024,211 "commit_id": "653305b998dd033365576db303b32dd5df3a6c54",212 "repo": "core",213 "path": "homeassistant/components/plex/media_browser.py",214 "file_name": "media_browser.py",215 "fun_name": "library_section_payload",216 "commit_message": "Support multiple Plex servers in media browser (#68321)",217 "code": "def library_section_payload(section):\n \n try:\n children_media_class = ITEM_TYPE_MEDIA_CLASS[section.TYPE]\n except KeyError as err:\n raise UnknownMediaType(f\"Unknown type received: {section.TYPE}\") from err\n server_id = section._server.machineIdentifier # pylint: disable=protected-access\n return BrowseMedia(\n title=section.title,\n media_class=MEDIA_CLASS_DIRECTORY,\n media_content_id=generate_plex_uri(server_id, section.key),\n media_content_type=\"library\",\n can_play=False,\n can_expand=True,\n children_media_class=children_media_class,\n )\n\n",218 "url": "https://github.com/home-assistant/core.git",219 "language": "Python",220 "ast_errors": "",221 "n_ast_errors": 0,222 "ast_levels": 13,223 "n_whitespaces": 116,224 "n_words": 34,225 "vocab_size": 33,226 "complexity": 2,227 "nloc": 15,228 "token_counts": 77,229 "n_ast_nodes": 126,230 "n_identifiers": 21231 },232 {233 "id": 213030,234 "commit_id": "a5db070f446b7cfebdaa6ad2e3dcf78f6105a272",235 "repo": "serverless-application-model",236 "path": "samtranslator/open_api/open_api.py",237 "file_name": "open_api.py",238 "fun_name": "gen_skeleton",239 "commit_message": "fix: Py27hash fix (#2182)\n\n* Add third party py27hash code\r\n\r\n* Add Py27UniStr and unit tests\r\n\r\n* Add py27hash_fix utils and tests\r\n\r\n* Add to_py27_compatible_template and tests\r\n\r\n* Apply py27hash fix to wherever it is needed\r\n\r\n* Apply py27hash fix, all tests pass except api_with_any_method_in_swagger\r\n\r\n* apply py27hash fix in openapi + run black\r\n\r\n* remove py27 testing\r\n\r\n* remove other py27 references\r\n\r\n* black fixes\r\n\r\n* fixes/typos\r\n\r\n* remove py27 from tox.ini\r\n\r\n* refactoring\r\n\r\n* third party notice\r\n\r\n* black\r\n\r\n* Fix py27hash fix to deal with null events\r\n\r\n* Fix Py27UniStr repr for unicode literals\r\n\r\n* black reformat\r\n\r\n* Update _template_has_api_resource to check data type more defensively\r\n\r\n* Apply py27Dict in _get_authorizers\r\n\r\n* Apply Py27Dict to authorizers and gateway responses which will go into swagger\r\n\r\n* Update to_py27_compatible_template to handle parameter_values; Add Py27LongInt class\r\n\r\n* Rename _convert_to_py27_dict to _convert_to_py27_type\r\n\r\n* Apply Py27UniStr to path param name\r\n\r\n* Handle HttpApi resource under to_py27_compatible_template\r\n\r\n* Fix InvalidDocumentException to not sort different exceptions\r\n\r\n* black reformat\r\n\r\n* Remove unnecessary test files\r\n\r\nCo-authored-by: Wing Fung Lau <4760060+hawflau@users.noreply.github.com>",240 "code": "def gen_skeleton():\n \n # create as Py27Dict and insert key one by one to preserve input order\n skeleton = Py27Dict()\n skeleton[\"openapi\"] = \"3.0.1\"\n skeleton[\"info\"] = Py27Dict()\n skeleton[\"info\"][\"version\"] = \"1.0\"\n skeleton[\"info\"][\"title\"] = ref(\"AWS::StackName\")\n skeleton[\"paths\"] = Py27Dict()\n return skeleton\n",241 "url": "https://github.com/aws/serverless-application-model.git",242 "language": "Python",243 "ast_errors": "",244 "n_ast_errors": 0,245 "ast_levels": 9,246 "n_whitespaces": 99,247 "n_words": 36,248 "vocab_size": 27,249 "complexity": 1,250 "nloc": 8,251 "token_counts": 55,252 "n_ast_nodes": 111,253 "n_identifiers": 4254 },255 {256 "id": 260625,257 "commit_id": "3312bc2ea6aad559643a1d920e3380fa123f627c",258 "repo": "scikit-learn",259 "path": "sklearn/decomposition/_kernel_pca.py",260 "file_name": "_kernel_pca.py",261 "fun_name": "fit",262 "commit_message": "MAINT validate parameter in KernelPCA (#24020)\n\nCo-authored-by: Julien Jerphanion <git@jjerphan.xyz>\r\nCo-authored-by: jeremiedbb <jeremiedbb@yahoo.fr>",263 "code": "def fit(self, X, y=None):\n \n self._validate_params()\n\n if self.fit_inverse_transform and self.kernel == \"precomputed\":\n raise ValueError(\"Cannot fit_inverse_transform with a precomputed kernel.\")\n X = self._validate_data(X, accept_sparse=\"csr\", copy=self.copy_X)\n self._centerer = KernelCenterer()\n K = self._get_kernel(X)\n self._fit_transform(K)\n\n if self.fit_inverse_transform:\n # no need to use the kernel to transform X, use shortcut expression\n X_transformed = self.eigenvectors_ * np.sqrt(self.eigenvalues_)\n\n self._fit_inverse_transform(X_transformed, X)\n\n self.X_fit_ = X\n return self\n",264 "url": "https://github.com/scikit-learn/scikit-learn.git",265 "language": "Python",266 "ast_errors": "",267 "n_ast_errors": 0,268 "ast_levels": 12,269 "n_whitespaces": 171,270 "n_words": 57,271 "vocab_size": 48,272 "complexity": 4,273 "nloc": 13,274 "token_counts": 106,275 "n_ast_nodes": 175,276 "n_identifiers": 24277 },278 {279 "id": 42787,280 "commit_id": "60eb9e106f5915398eafd6aa339ec710c102dc09",281 "repo": "airflow",282 "path": "airflow/providers/cncf/kubernetes/hooks/kubernetes.py",283 "file_name": "kubernetes.py",284 "fun_name": "get_conn",285 "commit_message": "Use KubernetesHook to create api client in KubernetesPodOperator (#20578)\n\nAdd support for k8s hook in KPO; use it always (even when no conn id); continue to consider the core k8s settings that KPO already takes into account but emit deprecation warning about them.\r\n\r\nKPO historically takes into account a few settings from core airflow cfg (e.g. verify ssl, tcp keepalive, context, config file, and in_cluster). So to use the hook to generate the client, somehow the hook has to take these settings into account. But we don't want the hook to consider these settings in general. So we read them in KPO and if necessary patch the hook and warn.",286 "code": "def get_conn(self) -> Any:\n \n\n in_cluster = self._coalesce_param(\n self.in_cluster, self.conn_extras.get(\"extra__kubernetes__in_cluster\") or None\n )\n cluster_context = self._coalesce_param(\n self.cluster_context, self.conn_extras.get(\"extra__kubernetes__cluster_context\") or None\n )\n kubeconfig_path = self._coalesce_param(\n self.config_file, self.conn_extras.get(\"extra__kubernetes__kube_config_path\") or None\n )\n\n kubeconfig = self.conn_extras.get(\"extra__kubernetes__kube_config\") or None\n num_selected_configuration = len([o for o in [in_cluster, kubeconfig, kubeconfig_path] if o])\n\n if num_selected_configuration > 1:\n raise AirflowException(\n \"Invalid connection configuration. Options kube_config_path, \"\n \"kube_config, in_cluster are mutually exclusive. \"\n \"You can only use one option at a time.\"\n )\n\n disable_verify_ssl = self._coalesce_param(\n self.disable_verify_ssl, _get_bool(self._get_field(\"disable_verify_ssl\"))\n )\n disable_tcp_keepalive = self._coalesce_param(\n self.disable_tcp_keepalive, _get_bool(self._get_field(\"disable_tcp_keepalive\"))\n )\n\n # BEGIN apply settings from core kubernetes configuration\n # this section should be removed in next major release\n deprecation_warnings: List[Tuple[str, Any]] = []\n if disable_verify_ssl is None and self._deprecated_core_disable_verify_ssl is True:\n deprecation_warnings.append(('verify_ssl', False))\n disable_verify_ssl = self._deprecated_core_disable_verify_ssl\n # by default, hook will try in_cluster first. so we only need to\n # apply core airflow config and alert when False and in_cluster not otherwise set.\n if in_cluster is None and self._deprecated_core_in_cluster is False:\n deprecation_warnings.append(('in_cluster', self._deprecated_core_in_cluster))\n in_cluster = self._deprecated_core_in_cluster\n if not cluster_context and self._deprecated_core_cluster_context:\n deprecation_warnings.append(('cluster_context', self._deprecated_core_cluster_context))\n cluster_context = self._deprecated_core_cluster_context\n if not kubeconfig_path and self._deprecated_core_config_file:\n deprecation_warnings.append(('config_file', self._deprecated_core_config_file))\n kubeconfig_path = self._deprecated_core_config_file\n if disable_tcp_keepalive is None and self._deprecated_core_disable_tcp_keepalive is True:\n deprecation_warnings.append(('enable_tcp_keepalive', False))\n disable_tcp_keepalive = True\n if deprecation_warnings:\n self._deprecation_warning_core_param(deprecation_warnings)\n # END apply settings from core kubernetes configuration\n\n if disable_verify_ssl is True:\n _disable_verify_ssl()\n if disable_tcp_keepalive is not True:\n _enable_tcp_keepalive()\n\n if in_cluster:\n self.log.debug(\"loading kube_config from: in_cluster configuration\")\n config.load_incluster_config()\n return client.ApiClient()\n\n if kubeconfig_path is not None:\n self.log.debug(\"loading kube_config from: %s\", kubeconfig_path)\n config.load_kube_config(\n config_file=kubeconfig_path,\n client_configuration=self.client_configuration,\n context=cluster_context,\n )\n return client.ApiClient()\n\n if kubeconfig is not None:\n with tempfile.NamedTemporaryFile() as temp_config:\n self.log.debug(\"loading kube_config from: connection kube_config\")\n temp_config.write(kubeconfig.encode())\n temp_config.flush()\n config.load_kube_config(\n config_file=temp_config.name,\n client_configuration=self.client_configuration,\n context=cluster_context,\n )\n return client.ApiClient()\n\n return self._get_default_client(cluster_context=cluster_context)\n",287 "url": "https://github.com/apache/airflow.git",288 "language": "Python",289 "ast_errors": "",290 "n_ast_errors": 0,291 "ast_levels": 13,292 "n_whitespaces": 1032,293 "n_words": 267,294 "vocab_size": 146,295 "complexity": 24,296 "nloc": 71,297 "token_counts": 460,298 "n_ast_nodes": 759,299 "n_identifiers": 49300 },301 {302 "id": 149758,303 "commit_id": "fc837c4daa27a18ff0e86128f4d52089b88fa5fb",304 "repo": "freqtrade",305 "path": "freqtrade/freqai/data_handler.py",306 "file_name": "data_handler.py",307 "fun_name": "load_data",308 "commit_message": "add freqao backend machinery, user interface, documentation",309 "code": "def load_data(self) -> Any:\n \n model = load(self.model_path+self.model_filename+\"_model.joblib\")\n\n with open(self.model_path+self.model_filename+\"_metadata.json\", 'r') as fp:\n self.data = json.load(fp)\n if self.data.get('training_features_list'):\n self.training_features_list = [*self.data.get('training_features_list')]\n\n self.data_dictionary['train_features'] = pd.read_pickle(self.model_path+\n self.model_filename+\"_trained_df.pkl\")\n\n self.model_path = self.data['model_path']\n self.model_filename = self.data['model_filename']\n if self.config['freqai']['feature_parameters']['principal_component_analysis']:\n self.pca = pk.load(open(self.model_path+self.model_filename+\"_pca_object.pkl\",\"rb\"))\n\n return model\n",310 "url": "https://github.com/freqtrade/freqtrade.git",311 "language": "Python",312 "ast_errors": "",313 "n_ast_errors": 0,314 "ast_levels": 15,315 "n_whitespaces": 180,316 "n_words": 37,317 "vocab_size": 29,318 "complexity": 3,319 "nloc": 18,320 "token_counts": 155,321 "n_ast_nodes": 272,322 "n_identifiers": 19323 },324 {325 "id": 144300,326 "commit_id": "c065e3f69ec248383d98b45a8d1c00832ccfdd57",327 "repo": "ray",328 "path": "python/ray/actor.py",329 "file_name": "actor.py",330 "fun_name": "_bind",331 "commit_message": "[Ray DAG] Implement experimental Ray DAG API for task/class (#22058)",332 "code": "def _bind(self, *args, **kwargs):\n \n from ray.experimental.dag.class_node import ClassNode\n\n return ClassNode(self.__ray_metadata__.modified_class, args, kwargs, {})\n\n",333 "url": "https://github.com/ray-project/ray.git",334 "language": "Python",335 "ast_errors": "",336 "n_ast_errors": 0,337 "ast_levels": 9,338 "n_whitespaces": 34,339 "n_words": 13,340 "vocab_size": 13,341 "complexity": 1,342 "nloc": 3,343 "token_counts": 38,344 "n_ast_nodes": 56,345 "n_identifiers": 11346 },347 {348 "id": 337524,349 "commit_id": "f56f4441b3d448f4a81d5131c03e7dd73eac3ba0",350 "repo": "accelerate",351 "path": "src/accelerate/utils/operations.py",352 "file_name": "operations.py",353 "fun_name": "find_device",354 "commit_message": "Big model inference (#345)\n\n* Big model inference\r\n\r\n* Reorganize port cleanup\r\n\r\n* Last cleanup\r\n\r\n* Test fix\r\n\r\n* Quality\r\n\r\n* Update src/accelerate/big_modeling.py\r\n\r\nCo-authored-by: Patrick von Platen <patrick.v.platen@gmail.com>\r\n\r\n* Fix bug in default mem\r\n\r\n* Check device map is complete\r\n\r\n* More tests\r\n\r\n* Make load function more general\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: Zachary Mueller <muellerzr@gmail.com>\r\n\r\n* Quality\r\n\r\n* Address more review comments\r\n\r\n* Check generation results for gpt2\r\n\r\n* Add main wrapper around everything\r\n\r\n* Tests for final API\r\n\r\n* Clean infer_auto_device\r\n\r\n* Type annotations\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: Sourab Mangrulkar <13534540+pacman100@users.noreply.github.com>\r\nCo-authored-by: Lysandre Debut <lysandre.debut@reseau.eseo.fr>\r\n\r\n* Address review comments\r\n\r\n* Last review comment for now\r\n\r\n* Fix bug in clean_device_map\r\n\r\n* Add doc\r\n\r\n* Style\r\n\r\n* Fixes + dtype support\r\n\r\n* Fix test\r\n\r\n* Add option to offload CPU state_dict\r\n\r\n* Indent typo\r\n\r\n* Final tweaks\r\n\r\nCo-authored-by: Patrick von Platen <patrick.v.platen@gmail.com>\r\nCo-authored-by: Zachary Mueller <muellerzr@gmail.com>\r\nCo-authored-by: Sourab Mangrulkar <13534540+pacman100@users.noreply.github.com>\r\nCo-authored-by: Lysandre Debut <lysandre.debut@reseau.eseo.fr>",355 "code": "def find_device(data):\n \n if isinstance(data, Mapping):\n for obj in data.values():\n device = find_device(obj)\n if device is not None:\n return device\n elif isinstance(data, (tuple, list)):\n for obj in data:\n device = find_device(obj)\n if device is not None:\n return device\n elif isinstance(data, torch.Tensor):\n return data.device\n",356 "url": "https://github.com/huggingface/accelerate.git",357 "language": "Python",358 "ast_errors": "",359 "n_ast_errors": 0,360 "ast_levels": 13,361 "n_whitespaces": 149,362 "n_words": 42,363 "vocab_size": 22,364 "complexity": 8,365 "nloc": 13,366 "token_counts": 82,367 "n_ast_nodes": 128,368 "n_identifiers": 11369 },370 {371 "id": 60126,372 "commit_id": "a368874d1b145c1ec5201e5efd3c26ce7c1e8611",373 "repo": "prefect",374 "path": "src/prefect/_internal/concurrency/primitives.py",375 "file_name": "primitives.py",376 "fun_name": "wait",377 "commit_message": "Add thread-safe async primitives `Event` and `Future` (#7865)\n\nCo-authored-by: Serina Grill <42048900+serinamarie@users.noreply.github.com>",378 "code": "async def wait(self) -> None:\n \n if self._is_set:\n return\n\n if not self._loop:\n self._loop = get_running_loop()\n self._event = asyncio.Event()\n\n await self._event.wait()\n",379 "url": "https://github.com/PrefectHQ/prefect.git",380 "language": "Python",381 "ast_errors": "",382 "n_ast_errors": 0,383 "ast_levels": 10,384 "n_whitespaces": 80,385 "n_words": 19,386 "vocab_size": 17,387 "complexity": 3,388 "nloc": 12,389 "token_counts": 44,390 "n_ast_nodes": 78,391 "n_identifiers": 8392 },393 {394 "id": 78295,395 "commit_id": "d967eccef28ce47f60d26be1c28f2d83a25f40b0",396 "repo": "wagtail",397 "path": "wagtail/contrib/settings/tests/generic/test_templates.py",398 "file_name": "test_templates.py",399 "fun_name": "test_get_settings_no_request",400 "commit_message": "Add generic settings to compliment site-specific settings (#8327)",401 "code": "def test_get_settings_no_request(self):\n \n context = Context()\n\n template = Template(\n \"{% load wagtailsettings_tags %}\"\n \"{% get_settings %}\"\n \"{{ settings.tests.testgenericsetting.title }}\"\n )\n\n self.assertEqual(template.render(context), self.default_settings.title)\n",402 "url": "https://github.com/wagtail/wagtail.git",403 "language": "Python",404 "ast_errors": "",405 "n_ast_errors": 0,406 "ast_levels": 10,407 "n_whitespaces": 89,408 "n_words": 21,409 "vocab_size": 18,410 "complexity": 1,411 "nloc": 8,412 "token_counts": 36,413 "n_ast_nodes": 67,414 "n_identifiers": 10415 },416 {417 "id": 208719,418 "commit_id": "dc5bcc1c50892a5128fcf128af28887226144927",419 "repo": "ipython",420 "path": "IPython/core/history.py",421 "file_name": "history.py",422 "fun_name": "get_tail",423 "commit_message": "This fixed the mixing of multiple history seen in #13631\n\nIt forces get_tail to put the current session last in the returned\nresults.",424 "code": "def get_tail(self, n=10, raw=True, output=False, include_latest=False):\n \n self.writeout_cache()\n if not include_latest:\n n += 1\n # cursor/line/entry\n this_cur = list(\n self._run_sql(\n \"WHERE session == ? ORDER BY line DESC LIMIT ? \",\n (self.session_number, n),\n raw=raw,\n output=output,\n )\n )\n other_cur = list(\n self._run_sql(\n \"WHERE session != ? ORDER BY session DESC, line DESC LIMIT ?\",\n (self.session_number, n),\n raw=raw,\n output=output,\n )\n )\n\n everything = this_cur + other_cur\n\n everything = everything[:n]\n\n if not include_latest:\n return list(everything)[:0:-1]\n return list(everything)[::-1]\n",425 "url": "https://github.com/ipython/ipython.git",426 "language": "Python",427 "ast_errors": "",428 "n_ast_errors": 0,429 "ast_levels": 12,430 "n_whitespaces": 344,431 "n_words": 73,432 "vocab_size": 44,433 "complexity": 3,434 "nloc": 25,435 "token_counts": 128,436 "n_ast_nodes": 198,437 "n_identifiers": 13438 },439 {440 "id": 268911,441 "commit_id": "b96518a22bfd92a29811e507dec0b34248a8a3f5",442 "repo": "keras",443 "path": "keras/mixed_precision/loss_scale_optimizer_test.py",444 "file_name": "loss_scale_optimizer_test.py",445 "fun_name": "opt_combinations_only",446 "commit_message": "- Consolidate disparate test-related files into a single testing_infra folder.\n- Cleanup TODO related to removing testing infra as a dependency of the Keras target.\n- Standardize import naming: there is now only \"test_combinations\" for test combinations, and \"test_utils\" for utilities. The TF utilities module \"test_util\" is now always imported as \"tf_test_utils\" to avoid confusion.\n\nPiperOrigin-RevId: 426773173",447 "code": "def opt_combinations_only():\n \n experimental_opt_combinations = test_combinations.combine(\n mode='eager', opt_cls=optimizer_experimental.Optimizer)\n orig_opt_combination = test_combinations.combine(\n opt_cls=optimizer_v2.OptimizerV2)\n return experimental_opt_combinations + orig_opt_combination\n\n\n@tf_test_utils.with_control_flow_v2",448 "url": "https://github.com/keras-team/keras.git",449 "language": "Python",450 "ast_errors": "@tf_test_utils.with_control_flow_v2",451 "n_ast_errors": 1,452 "ast_levels": 10,453 "n_whitespaces": 29,454 "n_words": 16,455 "vocab_size": 12,456 "complexity": 1,457 "nloc": 6,458 "token_counts": 37,459 "n_ast_nodes": 70,460 "n_identifiers": 13461 },462 {463 "id": 148285,464 "commit_id": "0e6c042e29cbbe429d81c9c1af3c75c261f00980",465 "repo": "ray",466 "path": "python/ray/_private/thirdparty/pathspec/util.py",467 "file_name": "util.py",468 "fun_name": "_normalize_entries",469 "commit_message": "[Bugfix] fix invalid excluding of Black (#24042)\n\n- We should use `--force-exclude` when we pass code path explicitly https://black.readthedocs.io/en/stable/usage_and_configuration/the_basics.html?highlight=--force-exclude#command-line-options\r\n- Recover the files in `python/ray/_private/thirdparty` which has been formatted in the PR https://github.com/ray-project/ray/pull/21975 by mistake.",470 "code": "def _normalize_entries(entries, separators=None):\n\t\n\tnorm_files = {}\n\tfor entry in entries:\n\t\tnorm_files[normalize_file(entry.path, separators=separators)] = entry\n\treturn norm_files\n\n",471 "url": "https://github.com/ray-project/ray.git",472 "language": "Python",473 "ast_errors": "",474 "n_ast_errors": 0,475 "ast_levels": 12,476 "n_whitespaces": 11,477 "n_words": 16,478 "vocab_size": 13,479 "complexity": 2,480 "nloc": 5,481 "token_counts": 36,482 "n_ast_nodes": 57,483 "n_identifiers": 7484 },485 {486 "id": 209543,487 "commit_id": "08b1f9d67c8e716fd44036a027bdc90dcb9fcfdf",488 "repo": "scapy",489 "path": "scapy/layers/inet.py",490 "file_name": "inet.py",491 "fun_name": "overlap_frag",492 "commit_message": "E275 - Missing whitespace after keyword (#3711)\n\nCo-authored-by: Alexander Aring <alex.aring@gmail.com>\r\nCo-authored-by: Anmol Sarma <me@anmolsarma.in>\r\nCo-authored-by: antoine.torre <torreantoine1@gmail.com>\r\nCo-authored-by: Antoine Vacher <devel@tigre-bleu.net>\r\nCo-authored-by: Arnaud Ebalard <arno@natisbad.org>\r\nCo-authored-by: atlowl <86038305+atlowl@users.noreply.github.com>\r\nCo-authored-by: Brian Bienvenu <brian@bienvenu.id.au>\r\nCo-authored-by: Chris Packham <chris.packham@alliedtelesis.co.nz>\r\nCo-authored-by: CQ <cq674350529@163.com>\r\nCo-authored-by: Daniel Collins <kinap@users.noreply.github.com>\r\nCo-authored-by: Federico Maggi <federico.maggi@gmail.com>\r\nCo-authored-by: Florian Maury <florian.maury@ssi.gouv.fr>\r\nCo-authored-by: _Frky <3105926+Frky@users.noreply.github.com>\r\nCo-authored-by: g-mahieux <37588339+g-mahieux@users.noreply.github.com>\r\nCo-authored-by: gpotter2 <gabriel@potter.fr>\r\nCo-authored-by: Guillaume Valadon <guillaume@valadon.net>\r\nCo-authored-by: Hao Zheng <haozheng10@gmail.com>\r\nCo-authored-by: Haresh Khandelwal <hareshkhandelwal@gmail.com>\r\nCo-authored-by: Harri Hämäläinen <hhamalai@iki.fi>\r\nCo-authored-by: hecke <hecke@naberius.de>\r\nCo-authored-by: Jan Romann <jan.romann@gmail.com>\r\nCo-authored-by: Jan Sebechlebsky <sebechlebskyjan@gmail.com>\r\nCo-authored-by: jdiog0 <43411724+jdiog0@users.noreply.github.com>\r\nCo-authored-by: jockque <38525640+jockque@users.noreply.github.com>\r\nCo-authored-by: Julien Bedel <30991560+JulienBedel@users.noreply.github.com>\r\nCo-authored-by: Keith Scott <kscott@mitre.org>\r\nCo-authored-by: Kfir Gollan <kfir@drivenets.com>\r\nCo-authored-by: Lars Munch <lars@segv.dk>\r\nCo-authored-by: ldp77 <52221370+ldp77@users.noreply.github.com>\r\nCo-authored-by: Leonard Crestez <cdleonard@gmail.com>\r\nCo-authored-by: Marcel Patzlaff <mpatzlaff@benocs.com>\r\nCo-authored-by: Martijn Thé <martijnthe@users.noreply.github.com>\r\nCo-authored-by: Martine Lenders <authmillenon@gmail.com>\r\nCo-authored-by: Michael Farrell <micolous+git@gmail.com>\r\nCo-authored-by: Michał Mirosław <mirq-linux@rere.qmqm.pl>\r\nCo-authored-by: mkaliszan <mkaliszan@benocs.com>\r\nCo-authored-by: mtury <maxence.tury@ssi.gouv.fr>\r\nCo-authored-by: Neale Ranns <nranns@cisco.com>\r\nCo-authored-by: Octavian Toader <Octavian.Toader@belden.com>\r\nCo-authored-by: Peter Eisenlohr <peter@eisenlohr.org>\r\nCo-authored-by: Phil <phil@secdev.org>\r\nCo-authored-by: Pierre Lalet <pierre@droids-corp.org>\r\nCo-authored-by: Pierre Lorinquer <pierre.lorinquer@ssi.gouv.fr>\r\nCo-authored-by: piersoh <42040737+piersoh@users.noreply.github.com>\r\nCo-authored-by: plorinquer <pierre.lorinquer@ssi.gouv.fr>\r\nCo-authored-by: pvinci <pvinci@users.noreply.github.com>\r\nCo-authored-by: Rahul Jadhav <nyrahul@gmail.com>\r\nCo-authored-by: Robin Jarry <robin.jarry@6wind.com>\r\nCo-authored-by: romain-perez <51962832+romain-perez@users.noreply.github.com>\r\nCo-authored-by: rperez <rperez@debian>\r\nCo-authored-by: Sabrina Dubroca <sd@queasysnail.net>\r\nCo-authored-by: Sebastian Baar <sebastian.baar@gmx.de>\r\nCo-authored-by: sebastien mainand <sebastien.mainand@ssi.gouv.fr>\r\nCo-authored-by: smehner1 <smehner1@gmail.com>\r\nCo-authored-by: speakinghedge <hecke@naberius.de>\r\nCo-authored-by: Steven Van Acker <steven@singularity.be>\r\nCo-authored-by: Thomas Faivre <thomas.faivre@6wind.com>\r\nCo-authored-by: Tran Tien Dat <peter.trantiendat@gmail.com>\r\nCo-authored-by: Wael Mahlous <wael.mahlous@gmail.com>\r\nCo-authored-by: waeva <74464394+waeva@users.noreply.github.com>\r\n\r\nCo-authored-by: Alexander Aring <alex.aring@gmail.com>\r\nCo-authored-by: Anmol Sarma <me@anmolsarma.in>\r\nCo-authored-by: antoine.torre <torreantoine1@gmail.com>\r\nCo-authored-by: Antoine Vacher <devel@tigre-bleu.net>\r\nCo-authored-by: Arnaud Ebalard <arno@natisbad.org>\r\nCo-authored-by: atlowl <86038305+atlowl@users.noreply.github.com>\r\nCo-authored-by: Brian Bienvenu <brian@bienvenu.id.au>\r\nCo-authored-by: Chris Packham <chris.packham@alliedtelesis.co.nz>\r\nCo-authored-by: CQ <cq674350529@163.com>\r\nCo-authored-by: Daniel Collins <kinap@users.noreply.github.com>\r\nCo-authored-by: Federico Maggi <federico.maggi@gmail.com>\r\nCo-authored-by: Florian Maury <florian.maury@ssi.gouv.fr>\r\nCo-authored-by: _Frky <3105926+Frky@users.noreply.github.com>\r\nCo-authored-by: g-mahieux <37588339+g-mahieux@users.noreply.github.com>\r\nCo-authored-by: gpotter2 <gabriel@potter.fr>\r\nCo-authored-by: Guillaume Valadon <guillaume@valadon.net>\r\nCo-authored-by: Hao Zheng <haozheng10@gmail.com>\r\nCo-authored-by: Haresh Khandelwal <hareshkhandelwal@gmail.com>\r\nCo-authored-by: Harri Hämäläinen <hhamalai@iki.fi>\r\nCo-authored-by: hecke <hecke@naberius.de>\r\nCo-authored-by: Jan Romann <jan.romann@gmail.com>\r\nCo-authored-by: Jan Sebechlebsky <sebechlebskyjan@gmail.com>\r\nCo-authored-by: jdiog0 <43411724+jdiog0@users.noreply.github.com>\r\nCo-authored-by: jockque <38525640+jockque@users.noreply.github.com>\r\nCo-authored-by: Julien Bedel <30991560+JulienBedel@users.noreply.github.com>\r\nCo-authored-by: Keith Scott <kscott@mitre.org>\r\nCo-authored-by: Kfir Gollan <kfir@drivenets.com>\r\nCo-authored-by: Lars Munch <lars@segv.dk>\r\nCo-authored-by: ldp77 <52221370+ldp77@users.noreply.github.com>\r\nCo-authored-by: Leonard Crestez <cdleonard@gmail.com>\r\nCo-authored-by: Marcel Patzlaff <mpatzlaff@benocs.com>\r\nCo-authored-by: Martijn Thé <martijnthe@users.noreply.github.com>\r\nCo-authored-by: Martine Lenders <authmillenon@gmail.com>\r\nCo-authored-by: Michael Farrell <micolous+git@gmail.com>\r\nCo-authored-by: Michał Mirosław <mirq-linux@rere.qmqm.pl>\r\nCo-authored-by: mkaliszan <mkaliszan@benocs.com>\r\nCo-authored-by: mtury <maxence.tury@ssi.gouv.fr>\r\nCo-authored-by: Neale Ranns <nranns@cisco.com>\r\nCo-authored-by: Octavian Toader <Octavian.Toader@belden.com>\r\nCo-authored-by: Peter Eisenlohr <peter@eisenlohr.org>\r\nCo-authored-by: Phil <phil@secdev.org>\r\nCo-authored-by: Pierre Lalet <pierre@droids-corp.org>\r\nCo-authored-by: Pierre Lorinquer <pierre.lorinquer@ssi.gouv.fr>\r\nCo-authored-by: piersoh <42040737+piersoh@users.noreply.github.com>\r\nCo-authored-by: pvinci <pvinci@users.noreply.github.com>\r\nCo-authored-by: Rahul Jadhav <nyrahul@gmail.com>\r\nCo-authored-by: Robin Jarry <robin.jarry@6wind.com>\r\nCo-authored-by: romain-perez <51962832+romain-perez@users.noreply.github.com>\r\nCo-authored-by: rperez <rperez@debian>\r\nCo-authored-by: Sabrina Dubroca <sd@queasysnail.net>\r\nCo-authored-by: Sebastian Baar <sebastian.baar@gmx.de>\r\nCo-authored-by: sebastien mainand <sebastien.mainand@ssi.gouv.fr>\r\nCo-authored-by: smehner1 <smehner1@gmail.com>\r\nCo-authored-by: Steven Van Acker <steven@singularity.be>\r\nCo-authored-by: Thomas Faivre <thomas.faivre@6wind.com>\r\nCo-authored-by: Tran Tien Dat <peter.trantiendat@gmail.com>\r\nCo-authored-by: Wael Mahlous <wael.mahlous@gmail.com>\r\nCo-authored-by: waeva <74464394+waeva@users.noreply.github.com>",493 "code": "def overlap_frag(p, overlap, fragsize=8, overlap_fragsize=None):\n \n\n if overlap_fragsize is None:\n overlap_fragsize = fragsize\n q = p.copy()\n del q[IP].payload\n q[IP].add_payload(overlap)\n\n qfrag = fragment(q, overlap_fragsize)\n qfrag[-1][IP].flags |= 1\n return qfrag + fragment(p, fragsize)\n\n",494 "url": "https://github.com/secdev/scapy.git",495 "language": "Python",496 "ast_errors": "",497 "n_ast_errors": 0,498 "ast_levels": 10,499 "n_whitespaces": 61,500 "n_words": 30,501 "vocab_size": 26,502 "complexity": 2,503 "nloc": 9,504 "token_counts": 76,505 "n_ast_nodes": 117,506 "n_identifiers": 13507 },508 {509 "id": 144233,510 "commit_id": "3f03ef8ba8016b095c611c4d2e118771e4a750ca",511 "repo": "ray",512 "path": "rllib/agents/alpha_star/distributed_learners.py",513 "file_name": "distributed_learners.py",514 "fun_name": "__len__",515 "commit_message": "[RLlib] AlphaStar: Parallelized, multi-agent/multi-GPU learning via league-based self-play. (#21356)",516 "code": "def __len__(self):\n \n return sum(len(s) for s in self.shards)\n",517 "url": "https://github.com/ray-project/ray.git",518 "language": "Python",519 "ast_errors": "",520 "n_ast_errors": 0,521 "ast_levels": 9,522 "n_whitespaces": 22,523 "n_words": 8,524 "vocab_size": 8,525 "complexity": 2,526 "nloc": 2,527 "token_counts": 20,528 "n_ast_nodes": 34,529 "n_identifiers": 6530 },531 {532 "id": 263805,533 "commit_id": "41483cb9e6d5086416c8fea6ad6781782c091c60",534 "repo": "pyinstaller",535 "path": "PyInstaller/utils/win32/winutils.py",536 "file_name": "winutils.py",537 "fun_name": "update_exe_pe_checksum",538 "commit_message": "winutils: optimize PE headers fixup\n\nAttempt to optimize PE headers fix-up from both time- and memory-\nintensity perspective.\n\nFirst, avoid specifying `fast_load=False` in `pefile.PE` constructor,\nbecause that triggers the bytes statistics collection\nhttps://github.com/erocarrera/pefile/blob/v2022.5.30/pefile.py#L2862-L2876\nwhich takes a long time for large files. Instead, we can obtain\nfull headers (required for build timestamp modification) by\ncalling `pe.full_load()` ourselves.\n\nSecond, use (an equivalent of) `MapFileAndCheckSumW` to compute\nthe PE checksum. For large files, it is orders of magnitude\nfaster than its pure-python `pefile.PE.generate_checksum`\ncounterpart.\n\nThe downside is that `MapFileAndCheckSumW` requires an on-disk\nfile as opposed to a memory buffer, so we need to split the\nPE headers fixup into two separate steps, with each modifying\nthe corresponding PE headers and (re)writing the whole file.\nEven so, this brings the fix-up process for a 700MB executable\ndown to seconds instead of minutes.\n\nIn addition, as noted on MSDN, `MapFileAndCheckSumW` internally\ncalls its ASCII variant (`MapFileAndCheckSumA`), so it cannot\nhandle file paths that contain characters that are not representable\nin the current code page. Therefore, we implement our own equivalent\nusing `ctypes` and pure widechar-based win32 API functions.",539 "code": "def update_exe_pe_checksum(exe_path):\n \n import pefile\n\n # Compute checksum using our equivalent of the MapFileAndCheckSumW - for large files, it is significantly faster\n # than pure-pyton pefile.PE.generate_checksum(). However, it requires the file to be on disk (i.e., cannot operate\n # on a memory buffer).\n try:\n checksum = compute_exe_pe_checksum(exe_path)\n except Exception as e:\n raise RuntimeError(\"Failed to compute PE checksum!\") from e\n\n # Update the checksum\n with pefile.PE(exe_path, fast_load=True) as pe:\n pe.OPTIONAL_HEADER.CheckSum = checksum\n\n # Generate updated EXE data\n data = pe.write()\n\n # Rewrite the exe\n with open(exe_path, 'wb') as fp:\n fp.write(data)\n\n",540 "url": "https://github.com/pyinstaller/pyinstaller.git",541 "language": "Python",542 "ast_errors": "",543 "n_ast_errors": 0,544 "ast_levels": 11,545 "n_whitespaces": 163,546 "n_words": 88,547 "vocab_size": 68,548 "complexity": 2,549 "nloc": 11,550 "token_counts": 72,551 "n_ast_nodes": 135,552 "n_identifiers": 17553 },554 {555 "id": 78323,556 "commit_id": "d967eccef28ce47f60d26be1c28f2d83a25f40b0",557 "repo": "wagtail",558 "path": "wagtail/contrib/settings/tests/site_specific/test_model.py",559 "file_name": "test_model.py",560 "fun_name": "test_get_page_url_when_for_settings_fetched_via_for_site",561 "commit_message": "Add generic settings to compliment site-specific settings (#8327)",562 "code": "def test_get_page_url_when_for_settings_fetched_via_for_site(self):\n \n self._create_importantpagessitesetting_object()\n\n settings = ImportantPagesSiteSetting.for_site(self.default_site)\n\n # Force site root paths query beforehand\n self.default_site.root_page._get_site_root_paths()\n\n for page_fk_field, expected_result in (\n (\"sign_up_page\", \"http://localhost/\"),\n (\"general_terms_page\", \"http://localhost/\"),\n (\"privacy_policy_page\", \"http://other/\"),\n ):\n with self.subTest(page_fk_field=page_fk_field):\n\n # only the first request for each URL will trigger queries.\n # 2 are triggered instead of 1 here, because tests use the\n # database cache backed, and the cache is queried each time\n # to fetch site root paths (because there's no 'request' to\n # store them on)\n\n with self.assertNumQueries(2):\n\n self.assertEqual(\n settings.get_page_url(page_fk_field), expected_result\n )\n\n # when called directly\n self.assertEqual(\n settings.get_page_url(page_fk_field), expected_result\n )\n\n # when called indirectly via shortcut\n self.assertEqual(\n getattr(settings.page_url, page_fk_field), expected_result\n )\n",563 "url": "https://github.com/wagtail/wagtail.git",564 "language": "Python",565 "ast_errors": "",566 "n_ast_errors": 0,567 "ast_levels": 16,568 "n_whitespaces": 506,569 "n_words": 102,570 "vocab_size": 74,571 "complexity": 2,572 "nloc": 20,573 "token_counts": 115,574 "n_ast_nodes": 201,575 "n_identifiers": 17576 },577 {578 "id": 269136,579 "commit_id": "e61cbc52fd3b0170769c120e9b8dabc8c4205322",580 "repo": "keras",581 "path": "keras/saving/saved_model/load.py",582 "file_name": "load.py",583 "fun_name": "recursively_deserialize_keras_object",584 "commit_message": "Support Keras saving/loading for ShardedVariables with arbitrary partitions.\n\nPiperOrigin-RevId: 439837516",585 "code": "def recursively_deserialize_keras_object(config, module_objects=None):\n \n if isinstance(config, dict):\n if 'class_name' in config:\n return generic_utils.deserialize_keras_object(\n config, module_objects=module_objects)\n else:\n return {\n key: recursively_deserialize_keras_object(config[key], module_objects)\n for key in config\n }\n elif isinstance(config, (tuple, list)):\n return [\n recursively_deserialize_keras_object(x, module_objects) for x in config\n ]\n else:\n raise ValueError(\n f'Unable to decode Keras layer config. Config should be a dictionary, '\n f'tuple or list. Received: config={config}')\n\n",586 "url": "https://github.com/keras-team/keras.git",587 "language": "Python",588 "ast_errors": "",589 "n_ast_errors": 0,590 "ast_levels": 15,591 "n_whitespaces": 140,592 "n_words": 58,593 "vocab_size": 48,594 "complexity": 6,595 "nloc": 18,596 "token_counts": 89,597 "n_ast_nodes": 142,598 "n_identifiers": 12599 },600 {601 "id": 309801,602 "commit_id": "dadcc5ebcbcf951ff677568b281c5897d990c8ae",603 "repo": "core",604 "path": "homeassistant/components/august/activity.py",605 "file_name": "activity.py",606 "fun_name": "get_latest_device_activity",607 "commit_message": "spelling: components/august (#64232)\n\nCo-authored-by: Josh Soref <jsoref@users.noreply.github.com>",608 "code": "def get_latest_device_activity(self, device_id, activity_types):\n \n if device_id not in self._latest_activities:\n return None\n\n latest_device_activities = self._latest_activities[device_id]\n latest_activity = None\n\n for activity_type in activity_types:\n if activity_type in latest_device_activities:\n if (\n latest_activity is not None\n and latest_device_activities[activity_type].activity_start_time\n <= latest_activity.activity_start_time\n ):\n continue\n latest_activity = latest_device_activities[activity_type]\n\n return latest_activity\n",609 "url": "https://github.com/home-assistant/core.git",610 "language": "Python",611 "ast_errors": "",612 "n_ast_errors": 0,613 "ast_levels": 14,614 "n_whitespaces": 227,615 "n_words": 42,616 "vocab_size": 28,617 "complexity": 6,618 "nloc": 15,619 "token_counts": 69,620 "n_ast_nodes": 106,621 "n_identifiers": 9622 },623 {624 "id": 159623,625 "commit_id": "9f634d248769198881bbb78ccd8d333982462ef5",626 "repo": "rasa",627 "path": "scripts/prepare_nightly_release.py",628 "file_name": "prepare_nightly_release.py",629 "fun_name": "project_root",630 "commit_message": "[ATO-114]Add nightly workflows and creation scripts",631 "code": "def project_root() -> Path:\n \n return Path(os.path.dirname(__file__)).parent.parent\n\n",632 "url": "https://github.com/RasaHQ/rasa.git",633 "language": "Python",634 "ast_errors": "",635 "n_ast_errors": 0,636 "ast_levels": 12,637 "n_whitespaces": 12,638 "n_words": 6,639 "vocab_size": 6,640 "complexity": 1,641 "nloc": 3,642 "token_counts": 23,643 "n_ast_nodes": 40,644 "n_identifiers": 7645 },646 {647 "id": 248026,648 "commit_id": "aa2811026402394b4013033f075d8f509cdc1257",649 "repo": "synapse",650 "path": "tests/storage/test_devices.py",651 "file_name": "test_devices.py",652 "fun_name": "add_device_change",653 "commit_message": "Process device list updates asynchronously (#12365)",654 "code": "def add_device_change(self, user_id, device_ids, host):\n \n\n for device_id in device_ids:\n stream_id = self.get_success(\n self.store.add_device_change_to_streams(\n \"user_id\", [device_id], [\"!some:room\"]\n )\n )\n\n self.get_success(\n self.store.add_device_list_outbound_pokes(\n user_id=user_id,\n device_id=device_id,\n room_id=\"!some:room\",\n stream_id=stream_id,\n hosts=[host],\n context={},\n )\n )\n",655 "url": "https://github.com/matrix-org/synapse.git",656 "language": "Python",657 "ast_errors": "",658 "n_ast_errors": 0,659 "ast_levels": 14,660 "n_whitespaces": 279,661 "n_words": 28,662 "vocab_size": 24,663 "complexity": 2,664 "nloc": 17,665 "token_counts": 79,666 "n_ast_nodes": 121,667 "n_identifiers": 14668 },669 {670 "id": 267050,671 "commit_id": "b9606417598217106e394c12c776d8c5ede9cd98",672 "repo": "ansible",673 "path": "test/lib/ansible_test/_internal/coverage_util.py",674 "file_name": "coverage_util.py",675 "fun_name": "self_check",676 "commit_message": "ansible-test - Support multiple coverage versions.\n\nci_complete\nci_coverage",677 "code": "def self_check() -> None:\n \n # Verify all supported Python versions have a coverage version.\n for version in SUPPORTED_PYTHON_VERSIONS:\n get_coverage_version(version)\n\n # Verify all controller Python versions are mapped to the latest coverage version.\n for version in CONTROLLER_PYTHON_VERSIONS:\n if get_coverage_version(version) != CONTROLLER_COVERAGE_VERSION:\n raise InternalError(f'Controller Python version {version} is not mapped to the latest coverage version.')\n\n\nself_check()\n",678 "url": "https://github.com/ansible/ansible.git",679 "language": "Python",680 "ast_errors": "",681 "n_ast_errors": 0,682 "ast_levels": 13,683 "n_whitespaces": 93,684 "n_words": 54,685 "vocab_size": 35,686 "complexity": 4,687 "nloc": 7,688 "token_counts": 35,689 "n_ast_nodes": 71,690 "n_identifiers": 7691 },692 {693 "id": 258546,694 "commit_id": "fb082b223dc9f1dd327f48dc9b830ee382d6f661",695 "repo": "scikit-learn",696 "path": "sklearn/neighbors/_regression.py",697 "file_name": "_regression.py",698 "fun_name": "predict",699 "commit_message": "MAINT Do not compute distances for uniform weighting (#22280)",700 "code": "def predict(self, X):\n \n if self.weights == \"uniform\":\n # In that case, we do not need the distances to perform\n # the weighting so we do not compute them.\n neigh_ind = self.kneighbors(X, return_distance=False)\n neigh_dist = None\n else:\n neigh_dist, neigh_ind = self.kneighbors(X)\n\n weights = _get_weights(neigh_dist, self.weights)\n\n _y = self._y\n if _y.ndim == 1:\n _y = _y.reshape((-1, 1))\n\n if weights is None:\n y_pred = np.mean(_y[neigh_ind], axis=1)\n else:\n y_pred = np.empty((X.shape[0], _y.shape[1]), dtype=np.float64)\n denom = np.sum(weights, axis=1)\n\n for j in range(_y.shape[1]):\n num = np.sum(_y[neigh_ind, j] * weights, axis=1)\n y_pred[:, j] = num / denom\n\n if self._y.ndim == 1:\n y_pred = y_pred.ravel()\n\n return y_pred\n\n",701 "url": "https://github.com/scikit-learn/scikit-learn.git",702 "language": "Python",703 "ast_errors": "",704 "n_ast_errors": 0,705 "ast_levels": 15,706 "n_whitespaces": 320,707 "n_words": 99,708 "vocab_size": 65,709 "complexity": 6,710 "nloc": 21,711 "token_counts": 199,712 "n_ast_nodes": 310,713 "n_identifiers": 26714 },715 {716 "id": 190825,717 "commit_id": "3c745ef193e9af9244cc406734e67815377472ed",718 "repo": "thumbor",719 "path": "thumbor/engines/extensions/pil.py",720 "file_name": "pil.py",721 "fun_name": "getImageDescriptor",722 "commit_message": "Reformat of files using black\n\nThese files were not properly formatted.",723 "code": "def getImageDescriptor(self, im, xy=None):\n \n\n # Defaule use full image and place at upper left\n if xy is None:\n xy = (0, 0)\n\n # Image separator,\n bb = b\"\\x2C\"\n\n # Image position and size\n bb += int2long(xy[0]) # Left position\n bb += int2long(xy[1]) # Top position\n bb += int2long(im.size[0]) # image width\n bb += int2long(im.size[1]) # image height\n\n # packed field: local color table flag1, interlace0, sorted table0,\n # reserved00, lct size111=7=2^(7+1)=256.\n\n bb += b\"\\x87\"\n\n # LZW minimum size code now comes later,\n # begining of [image data] blocks\n return bb\n",724 "url": "https://github.com/thumbor/thumbor.git",725 "language": "Python",726 "ast_errors": "",727 "n_ast_errors": 0,728 "ast_levels": 10,729 "n_whitespaces": 217,730 "n_words": 90,731 "vocab_size": 61,732 "complexity": 2,733 "nloc": 10,734 "token_counts": 74,735 "n_ast_nodes": 130,736 "n_identifiers": 7737 },738 {739 "id": 272348,740 "commit_id": "84afc5193d38057e2e2badf9c889ea87d80d8fbf",741 "repo": "keras",742 "path": "keras/layers/attention/attention_test.py",743 "file_name": "attention_test.py",744 "fun_name": "test_calculate_scores_one_dim_with_scale",745 "commit_message": "Reformatting the codebase with black.\n\nPiperOrigin-RevId: 450093126",746 "code": "def test_calculate_scores_one_dim_with_scale(self):\n \n # Query tensor of shape [1, 1, 1]\n q = np.array([[[1.1]]], dtype=np.float32)\n # Key tensor of shape [1, 1, 1]\n k = np.array([[[1.6]]], dtype=np.float32)\n attention_layer = keras.layers.Attention(use_scale=True)\n attention_layer.build(input_shape=([1, 1, 1], [1, 1, 1]))\n attention_layer.scale = -2.0\n actual = attention_layer._calculate_scores(query=q, key=k)\n\n # Expected tensor of shape [1, 1, 1].\n # expected000 = -2*1.1*1.6 = -3.52\n expected = np.array([[[-3.52]]], dtype=np.float32)\n self.assertAllClose(expected, actual)\n",747 "url": "https://github.com/keras-team/keras.git",748 "language": "Python",749 "ast_errors": "",750 "n_ast_errors": 0,751 "ast_levels": 12,752 "n_whitespaces": 153,753 "n_words": 62,754 "vocab_size": 36,755 "complexity": 1,756 "nloc": 9,757 "token_counts": 139,758 "n_ast_nodes": 203,759 "n_identifiers": 22760 },761 {762 "id": 281114,763 "commit_id": "ea964109d654394cc0a5237e6ec5510ba6404097",764 "repo": "OpenBBTerminal",765 "path": "gamestonk_terminal/cryptocurrency/cryptocurrency_helpers.py",766 "file_name": "cryptocurrency_helpers.py",767 "fun_name": "prepare_all_coins_df",768 "commit_message": "Crypto menu refactor (#1119)\n\n* enabled some crypto commands in dd to be called independent of source loaded\r\n\r\n* support for coin_map_df in all dd functions + load ta and plot chart refactor\r\n\r\n* updated tests and removed coingecko scrapping where possible\r\n\r\n* removed ref of command from hugo\r\n\r\n* updated pycoingecko version\r\n\r\n* refactoring load\r\n\r\n* refactored load to fetch prices; pred can run independent of source now\r\n\r\n* load by default usd on cp/cg and usdt on cb/bin\r\n\r\n* updated to rich for formatting and updated dependencies\r\n\r\n* fixed changes requested\r\n\r\n* update docs\r\n\r\n* revert discord requirements\r\n\r\n* removed absolute from calculate change for price\r\n\r\n* fixing pr issues\r\n\r\n* fix loading issue when similar coins exist, move coins to home, fill n/a\r\n\r\n* update docs for coins\r\n\r\n* adds load to ta and pred menu",769 "code": "def prepare_all_coins_df() -> pd.DataFrame:\n \n\n gecko_coins_df = load_coins_list(\"coingecko_coins.json\")\n\n paprika_coins_df = load_coins_list(\"coinpaprika_coins.json\")\n paprika_coins_df = paprika_coins_df[paprika_coins_df[\"is_active\"]]\n paprika_coins_df = paprika_coins_df[[\"rank\", \"id\", \"name\", \"symbol\", \"type\"]]\n\n # TODO: Think about scheduled job, that once a day will update data\n\n binance_coins_df = load_binance_map().rename(columns={\"symbol\": \"Binance\"})\n coinbase_coins_df = load_coinbase_map().rename(columns={\"symbol\": \"Coinbase\"})\n gecko_paprika_coins_df = pd.merge(\n gecko_coins_df, paprika_coins_df, on=\"name\", how=\"left\"\n )\n df_merged = pd.merge(\n left=gecko_paprika_coins_df,\n right=binance_coins_df,\n left_on=\"id_x\",\n right_on=\"id\",\n how=\"left\",\n )\n df_merged.rename(\n columns={\n \"id_x\": \"CoinGecko\",\n \"symbol_x\": \"Symbol\",\n \"id_y\": \"CoinPaprika\",\n },\n inplace=True,\n )\n\n df_merged = pd.merge(\n left=df_merged,\n right=coinbase_coins_df,\n left_on=\"CoinGecko\",\n right_on=\"id\",\n how=\"left\",\n )\n\n return df_merged[[\"CoinGecko\", \"CoinPaprika\", \"Binance\", \"Coinbase\", \"Symbol\"]]\n\n",770 "url": "https://github.com/OpenBB-finance/OpenBBTerminal.git",771 "language": "Python",772 "ast_errors": "",773 "n_ast_errors": 0,774 "ast_levels": 12,775 "n_whitespaces": 266,776 "n_words": 84,777 "vocab_size": 65,778 "complexity": 1,779 "nloc": 48,780 "token_counts": 191,781 "n_ast_nodes": 339,782 "n_identifiers": 22783 },784 {785 "id": 34009,786 "commit_id": "28e091430eea9e0d40839e56fd0d57aec262f5f9",787 "repo": "transformers",788 "path": "src/transformers/models/nystromformer/modeling_nystromformer.py",789 "file_name": "modeling_nystromformer.py",790 "fun_name": "_set_gradient_checkpointing",791 "commit_message": "Add Nystromformer (#14659)\n\n* Initial commit\r\n\r\n* Config and modelling changes\r\n\r\nAdded Nystromformer-specific attributes to config and removed all decoder functionality from modelling.\r\n\r\n* Modelling and test changes\r\n\r\nAdded Nystrom approximation and removed decoder tests.\r\n\r\n* Code quality fixes\r\n\r\n* Modeling changes and conversion script\r\n\r\nInitial commits to conversion script, modeling changes.\r\n\r\n* Minor modeling changes and conversion script\r\n\r\n* Modeling changes\r\n\r\n* Correct modeling, add tests and documentation\r\n\r\n* Code refactor\r\n\r\n* Remove tokenizers\r\n\r\n* Code refactor\r\n\r\n* Update __init__.py\r\n\r\n* Fix bugs\r\n\r\n* Update src/transformers/__init__.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/__init__.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/__init__.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update docs/source/model_doc/nystromformer.mdx\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/configuration_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/configuration_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/configuration_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/configuration_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/convert_nystromformer_original_pytorch_checkpoint_to_pytorch.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update src/transformers/models/nystromformer/configuration_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update modeling and test_modeling\r\n\r\n* Code refactor\r\n\r\n* .rst to .mdx\r\n\r\n* doc changes\r\n\r\n* Doc changes\r\n\r\n* Update modeling_nystromformer.py\r\n\r\n* Doc changes\r\n\r\n* Fix copies\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update configuration_nystromformer.py\r\n\r\n* Fix copies\r\n\r\n* Update tests/test_modeling_nystromformer.py\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\n\r\n* Update test_modeling_nystromformer.py\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: Lysandre Debut <lysandre@huggingface.co>\r\n\r\n* Fix code style\r\n\r\n* Update modeling_nystromformer.py\r\n\r\n* Update modeling_nystromformer.py\r\n\r\n* Fix code style\r\n\r\n* Reformat modeling file\r\n\r\n* Update modeling_nystromformer.py\r\n\r\n* Modify NystromformerForMultipleChoice\r\n\r\n* Fix code quality\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: Sylvain Gugger <35901082+sgugger@users.noreply.github.com>\r\n\r\n* Code style changes and torch.no_grad()\r\n\r\n* make style\r\n\r\n* Apply suggestions from code review\r\n\r\nCo-authored-by: NielsRogge <48327001+NielsRogge@users.noreply.github.com>\r\nCo-authored-by: Lysandre Debut <lysandre@huggingface.co>\r\nCo-authored-by: Sylvain Gugger <35901082+sgugger@users.noreply.github.com>",792 "code": "def _set_gradient_checkpointing(self, module, value=False):\n if isinstance(module, NystromformerEncoder):\n module.gradient_checkpointing = value\n\n\nNYSTROMFORMER_START_DOCSTRING = r\n\nNYSTROMFORMER_INPUTS_DOCSTRING = r\n\n\n@add_start_docstrings(\n \"The bare Nyströmformer Model transformer outputting raw hidden-states without any specific head on top.\",\n NYSTROMFORMER_START_DOCSTRING,\n)",793 "url": "https://github.com/huggingface/transformers.git",794 "language": "Python",795 "ast_errors": "@add_start_docstrings(\n \"The bare Nyströmformer Model transformer outputting raw hidden-states without any specific head on top.\",\n NYSTROMFORMER_START_DOCSTRING,\n)",796 "n_ast_errors": 1,797 "ast_levels": 9,798 "n_whitespaces": 52,799 "n_words": 33,800 "vocab_size": 30,801 "complexity": 2,802 "nloc": 3,803 "token_counts": 24,804 "n_ast_nodes": 64,805 "n_identifiers": 10806 },807 {808 "id": 119984,809 "commit_id": "3184dd65a222354bffa2466d9a375162f5649132",810 "repo": "jax",811 "path": "jax/experimental/sparse/bcoo.py",812 "file_name": "bcoo.py",813 "fun_name": "bcoo_dot_general_sampled",814 "commit_message": "[sparse] Update docstrings for bcoo primitives.\n\nPiperOrigin-RevId: 438685829",815 "code": "def bcoo_dot_general_sampled(A, B, indices, *, dimension_numbers):\n \n (lhs_contract, rhs_contract), (lhs_batch, rhs_batch) = dimension_numbers\n cdims = (api_util._ensure_index_tuple(lhs_contract),\n api_util._ensure_index_tuple(rhs_contract))\n bdims = (api_util._ensure_index_tuple(lhs_batch),\n api_util._ensure_index_tuple(rhs_batch))\n return bcoo_dot_general_sampled_p.bind(A, B, indices,\n dimension_numbers=(cdims, bdims))\n\n@bcoo_dot_general_sampled_p.def_impl",816 "url": "https://github.com/google/jax.git",817 "language": "Python",818 "ast_errors": "@bcoo_dot_general_sampled_p.def_impl",819 "n_ast_errors": 1,820 "ast_levels": 9,821 "n_whitespaces": 91,822 "n_words": 27,823 "vocab_size": 23,824 "complexity": 1,825 "nloc": 8,826 "token_counts": 80,827 "n_ast_nodes": 124,828 "n_identifiers": 16829 },830 {831 "id": 80736,832 "commit_id": "a3a216f91f1158fd54c001c34cbdf2f68ccbc272",833 "repo": "awx",834 "path": "awx/main/migrations/_inventory_source.py",835 "file_name": "_inventory_source.py",836 "fun_name": "_get_instance_id",837 "commit_message": "Fix up new Django 3.0 deprecations\n\nMostly text based: force/smart_text, ugettext_*",838 "code": "def _get_instance_id(from_dict, new_id, default=''):\n \n instance_id = default\n for key in new_id.split('.'):\n if not hasattr(from_dict, 'get'):\n instance_id = default\n break\n instance_id = from_dict.get(key, default)\n from_dict = instance_id\n return smart_str(instance_id)\n\n",839 "url": "https://github.com/ansible/awx.git",840 "language": "Python",841 "ast_errors": "",842 "n_ast_errors": 0,843 "ast_levels": 11,844 "n_whitespaces": 83,845 "n_words": 28,846 "vocab_size": 21,847 "complexity": 3,848 "nloc": 9,849 "token_counts": 56,850 "n_ast_nodes": 95,851 "n_identifiers": 10852 },853 {854 "id": 100396,855 "commit_id": "c1512fd41d86ef47a5d1ce618d6d755ef7cbacdf",856 "repo": "faceswap",857 "path": "plugins/train/trainer/_base.py",858 "file_name": "_base.py",859 "fun_name": "compile_sample",860 "commit_message": "Update code to support Tensorflow versions up to 2.8 (#1213)\n\n* Update maximum tf version in setup + requirements\r\n\r\n* - bump max version of tf version in launcher\r\n- standardise tf version check\r\n\r\n* update keras get_custom_objects for tf>2.6\r\n\r\n* bugfix: force black text in GUI file dialogs (linux)\r\n\r\n* dssim loss - Move to stock tf.ssim function\r\n\r\n* Update optimizer imports for compatibility\r\n\r\n* fix logging for tf2.8\r\n\r\n* Fix GUI graphing for TF2.8\r\n\r\n* update tests\r\n\r\n* bump requirements.txt versions\r\n\r\n* Remove limit on nvidia-ml-py\r\n\r\n* Graphing bugfixes\r\n - Prevent live graph from displaying if data not yet available\r\n\r\n* bugfix: Live graph. Collect loss labels correctly\r\n\r\n* fix: live graph - swallow inconsistent loss errors\r\n\r\n* Bugfix: Prevent live graph from clearing during training\r\n\r\n* Fix graphing for AMD",861 "code": "def compile_sample(self, batch_size, samples=None, images=None, masks=None):\n \n num_images = self._config.get(\"preview_images\", 14)\n num_images = min(batch_size, num_images) if batch_size is not None else num_images\n retval = {}\n for side in (\"a\", \"b\"):\n logger.debug(\"Compiling samples: (side: '%s', samples: %s)\", side, num_images)\n side_images = images[side] if images is not None else self._target[side]\n side_masks = masks[side] if masks is not None else self._masks[side]\n side_samples = samples[side] if samples is not None else self._samples[side]\n retval[side] = [side_samples[0:num_images],\n side_images[0:num_images],\n side_masks[0:num_images]]\n return retval\n",862 "url": "https://github.com/deepfakes/faceswap.git",863 "language": "Python",864 "ast_errors": "",865 "n_ast_errors": 0,866 "ast_levels": 11,867 "n_whitespaces": 225,868 "n_words": 74,869 "vocab_size": 48,870 "complexity": 6,871 "nloc": 13,872 "token_counts": 153,873 "n_ast_nodes": 225,874 "n_identifiers": 20875 },876 {877 "id": 314211,878 "commit_id": "90e1fb6ce2faadb9a35fdbe1774fce7b4456364f",879 "repo": "core",880 "path": "homeassistant/components/weather/__init__.py",881 "file_name": "__init__.py",882 "fun_name": "temperature",883 "commit_message": "Weather unit conversion (#73441)\n\nCo-authored-by: Erik <erik@montnemery.com>",884 "code": "def temperature(self) -> float | None:\n \n return self._attr_temperature\n",885 "url": "https://github.com/home-assistant/core.git",886 "language": "Python",887 "ast_errors": "",888 "n_ast_errors": 0,889 "ast_levels": 6,890 "n_whitespaces": 22,891 "n_words": 8,892 "vocab_size": 8,893 "complexity": 1,894 "nloc": 6,895 "token_counts": 14,896 "n_ast_nodes": 25,897 "n_identifiers": 4898 },899 {900 "id": 20801,901 "commit_id": "f3166e673fe8d40277b804d35d77dcdb760fc3b3",902 "repo": "pipenv",903 "path": "pipenv/patched/notpip/_vendor/rich/progress.py",904 "file_name": "progress.py",905 "fun_name": "get_time",906 "commit_message": "check point progress on only bringing in pip==22.0.4 (#4966)\n\n* vendor in pip==22.0.4\r\n\r\n* updating vendor packaging version\r\n\r\n* update pipdeptree to fix pipenv graph with new version of pip.\r\n\r\n* Vendoring of pip-shims 0.7.0\r\n\r\n* Vendoring of requirementslib 1.6.3\r\n\r\n* Update pip index safety restrictions patch for pip==22.0.4\r\n\r\n* Update patches\r\n\r\n* exclude pyptoject.toml from black to see if that helps.\r\n\r\n* Move this part of the hash collection back to the top (like prior implementation) because it affects the outcome of this test now in pip 22.0.4",907 "code": "def get_time(self) -> float:\n \n return self._get_time()\n",908 "url": "https://github.com/pypa/pipenv.git",909 "language": "Python",910 "ast_errors": "",911 "n_ast_errors": 0,912 "ast_levels": 7,913 "n_whitespaces": 20,914 "n_words": 6,915 "vocab_size": 6,916 "complexity": 1,917 "nloc": 3,918 "token_counts": 14,919 "n_ast_nodes": 26,920 "n_identifiers": 4921 },922 {923 "id": 118718,924 "commit_id": "2c153aa179a27539f856e389870161d5a58da213",925 "repo": "streamlit",926 "path": "lib/tests/streamlit/legacy_dataframe_styling_test.py",927 "file_name": "legacy_dataframe_styling_test.py",928 "fun_name": "test_add_unstyled_rows_to_styled_rows",929 "commit_message": "Pandas 1.4 styler fix (#4316)\n\nChange the way we detect custom styling in a DataFrame, to account for changes in Pandas 1.4.\r\n\r\nOur DataFrame styling support is based on internal Pandas APIs, so they're always subject to change out from underneath us. In general, we'd prefer to only pass `display_value` data to the frontend when a DataFrame cell has been custom-formatted by the user, to save on bandwidth. However, Panda's Styler's internals are private, and it doesn't give us a consistent way of testing whether a cell has a custom `display_value` or not. \r\n\r\nPrior to Pandas 1.4, we could test whether a cell's `display_value` differed from its `value`, and only stick the `display_value` in the protobuf when that was the case. In 1.4, an unmodified Styler will contain `display_value` strings for all cells, regardless of whether any formatting has been applied to that cell, so we no longer have this ability (or at least I couldn't figure out a reasonable way to test for this). \r\n\r\nSo instead, as of this PR, calling `st._legacy_dataframe(df.styler)` will *always* result in `display_value` strings being written to the dataframe protobuf (even though there isn't any custom formatting). This means that styled DataFrames may result in more data being sent to the frontend now than was the case before. In practice, I don't think this is a big deal - only the legacy DataFrame code has styling support; and often, if you're styling a DataFrame, you're customizing the formatting on most or all of its cells anyway.\r\n\r\nI also made a number of small type-safety changes as I was working with the dataframe code, and those are all in the PR as well. (I've left a PR comment under the actual logic changes.)",930 "code": "def test_add_unstyled_rows_to_styled_rows(self, st_element, get_proto):\n \n df1 = pd.DataFrame([5, 6])\n df2 = pd.DataFrame([7, 8])\n\n css_values = [\n {css_s(\"color\", \"black\")},\n {css_s(\"color\", \"black\")},\n set(),\n set(),\n ]\n\n x = st_element(df1.style.applymap(lambda val: \"color: black\"))\n\n x._legacy_add_rows(df2)\n\n proto_df = get_proto(self._get_element())\n self._assert_column_css_styles(proto_df, 0, css_values)\n",931 "url": "https://github.com/streamlit/streamlit.git",932 "language": "Python",933 "ast_errors": "",934 "n_ast_errors": 0,935 "ast_levels": 12,936 "n_whitespaces": 142,937 "n_words": 35,938 "vocab_size": 28,939 "complexity": 1,940 "nloc": 13,941 "token_counts": 106,942 "n_ast_nodes": 173,943 "n_identifiers": 19944 },945 {946 "id": 282770,947 "commit_id": "401e4c739a6f9d18944e0ab49c782e97b56fda94",948 "repo": "OpenBBTerminal",949 "path": "gamestonk_terminal/helper_funcs.py",950 "file_name": "helper_funcs.py",951 "fun_name": "handle_error_code",952 "commit_message": "Output Missing API Key Message to Console (#1357)\n\n* Decorator to output error msg to console of missing API Key\r\n\r\n* Refactor FMP & alpha advantage\r\n\r\n* Refactor FRED & QUANDL\r\n\r\n* Refactor Polygon\r\n\r\n* Refactor FRED\r\n\r\n* Refactor FRED\r\n\r\n* Refactor Finnhub & coinmarketcap & Newsapi\r\n\r\n* Allow disabling of check api\r\n\r\n* Updating tests : disable check api for tests\r\n\r\n* Refactor Finnhub & SI & Binance\r\n\r\n* Fix linting\r\n\r\n* Fix test & add black formatting\r\n\r\n* Fix test failing\r\n\r\n* Fix test failing\r\n\r\n* Refactor CryptoPanic & Whales alert & Glassnode & Coinglass\r\n\r\n* Refactor ETHexplorer & Smartstake & Alpha Advanage & Coinbase\r\n\r\n* Add decorators to controllers\r\n\r\n* Fix test & Refactor Coinbase, RH, Reddit\r\n\r\n* Add contributing guideline\r\n\r\n* Update CONTRIBUTING.md\r\n\r\n* Update CONTRIBUTING.md\r\n\r\n* fix tests\r\n\r\n* add decorator to snews cmd\r\n\r\nCo-authored-by: Chavithra PARANA <chavithra@gmail.com>\r\nCo-authored-by: didierlopes.eth <dro.lopes@campus.fct.unl.pt>",953 "code": "def handle_error_code(requests_obj, error_code_map):\n \n for error_code, error_msg in error_code_map.items():\n if requests_obj.status_code == error_code:\n console.print(error_msg)\n",954 "url": "https://github.com/OpenBB-finance/OpenBBTerminal.git",955 "language": "Python",956 "ast_errors": "",957 "n_ast_errors": 0,958 "ast_levels": 11,959 "n_whitespaces": 37,960 "n_words": 13,961 "vocab_size": 13,962 "complexity": 3,963 "nloc": 4,964 "token_counts": 32,965 "n_ast_nodes": 53,966 "n_identifiers": 9967 },968 {969 "id": 21882,970 "commit_id": "cd5a9683be69c86c8f3adcd13385a9bc5db198ec",971 "repo": "pipenv",972 "path": "pipenv/patched/pip/_vendor/chardet/__init__.py",973 "file_name": "__init__.py",974 "fun_name": "detect",975 "commit_message": "Rename notpip to pip. Vendor in pip-22.2.1 and latest requirementslib and vistir.",976 "code": "def detect(byte_str):\n \n if not isinstance(byte_str, bytearray):\n if not isinstance(byte_str, bytes):\n raise TypeError(\n f\"Expected object of type bytes or bytearray, got: {type(byte_str)}\"\n )\n byte_str = bytearray(byte_str)\n detector = UniversalDetector()\n detector.feed(byte_str)\n return detector.close()\n\n",977 "url": "https://github.com/pypa/pipenv.git",978 "language": "Python",979 "ast_errors": "",980 "n_ast_errors": 0,981 "ast_levels": 15,982 "n_whitespaces": 97,983 "n_words": 31,984 "vocab_size": 27,985 "complexity": 3,986 "nloc": 10,987 "token_counts": 53,988 "n_ast_nodes": 99,989 "n_identifiers": 11990 },991 {992 "id": 212923,993 "commit_id": "f776589349476a41b98aa1f467aff2f30e2a8fc2",994 "repo": "PySimpleGUI",995 "path": "PySimpleGUI.py",996 "file_name": "PySimpleGUI.py",997 "fun_name": "delete_file",998 "commit_message": "Added report_error setting for user_settings_delete_file. Global Settings window complete rework to use Tabs. Hoping nothing broke, but just remember things are in flux for a little bit while the ttk scrollbars are finishing up",999 "code": "def delete_file(self, filename=None, path=None, report_error=False):\n \n\n if filename is not None or path is not None or (filename is None and path is None):\n self.set_location(filename=filename, path=path)\n try:\n os.remove(self.full_filename)\n except Exception as e:\n if report_error:\n _error_popup_with_traceback('UserSettings delete_file warning ***', 'Exception trying to perform os.remove', e)\n self.dict = {}\n",1000 "url": "https://github.com/PySimpleGUI/PySimpleGUI.git",1001 "language": "Python",1002 "ast_errors": "",1003 "n_ast_errors": 0,1004 "ast_levels": 13,1005 "n_whitespaces": 129,1006 "n_words": 46,1007 "vocab_size": 37,1008 "complexity": 7,1009 "nloc": 9,1010 "token_counts": 83,1011 "n_ast_nodes": 133,1012 "n_identifiers": 131013 },1014 {1015 "id": 60235,1016 "commit_id": "cc4d0564756ca067516f71718a3d135996525909",1017 "repo": "transferlearning",1018 "path": "code/deep/BJMMD/caffe/python/caffe/coord_map.py",1019 "file_name": "coord_map.py",1020 "fun_name": "compose",1021 "commit_message": "Balanced joint maximum mean discrepancy for deep transfer learning",1022 "code": "def compose(base_map, next_map):\n \n ax1, a1, b1 = base_map\n ax2, a2, b2 = next_map\n if ax1 is None:\n ax = ax2\n elif ax2 is None or ax1 == ax2:\n ax = ax1\n else:\n raise AxisMismatchException\n return ax, a1 * a2, a1 * b2 + b1\n\n",1023 "url": "https://github.com/jindongwang/transferlearning.git",1024 "language": "Python",1025 "ast_errors": "",1026 "n_ast_errors": 0,1027 "ast_levels": 9,1028 "n_whitespaces": 86,1029 "n_words": 44,1030 "vocab_size": 31,1031 "complexity": 4,1032 "nloc": 10,1033 "token_counts": 58,1034 "n_ast_nodes": 91,1035 "n_identifiers": 111036 },1037 {1038 "id": 225809,1039 "commit_id": "c22d865acb3899a181921d94b6e94e665a12b432",1040 "repo": "llama_index",1041 "path": "gpt_index/schema.py",1042 "file_name": "schema.py",1043 "fun_name": "is_doc_id_none",1044 "commit_message": "Add index composability! (#86)\n\nSummary of changes\r\n- Bumped version to 0.1.0 \r\n- Abstracted out a BaseDocument class that both Document (from data loaders) and IndexStruct (our data struct classes) inherit from.\r\n- Add a DocumentStore that contains the id's of all BaseDocuments. Both Document objects and IndexStruct objects are registered in here, allowing us to recursively fetch and query sub-index structures within an index structure.\r\n- Add a reference document id to each Node class. This allows us to recursively query within another index struct after we traverse a node, if the reference document id of that node corresponds to another index struct in the DocumentStore.\r\n- Use Node as the central abstraction containing both \"text\" as well as a reference document_id: use for List, Tree, KeywordTable\r\n- Factored out a QueryRunner to recursively run queries. I grappled with some circular dependency issues but I believe the current approach works.\r\n- Add a bunch of unit tests\r\n\r\nCo-authored-by: Jerry Liu <jerry@robustintelligence.com>",1045 "code": "def is_doc_id_none(self) -> bool:\n \n return self.doc_id is None\n\n\n@dataclass",1046 "url": "https://github.com/jerryjliu/llama_index.git",1047 "language": "Python",1048 "ast_errors": "@dataclass",1049 "n_ast_errors": 1,1050 "ast_levels": 7,1051 "n_whitespaces": 22,1052 "n_words": 9,1053 "vocab_size": 9,1054 "complexity": 1,1055 "nloc": 3,1056 "token_counts": 14,1057 "n_ast_nodes": 29,1058 "n_identifiers": 51059 },1060 {1061 "id": 301395,1062 "commit_id": "42c80dda85f567192c182da2b4c603408a890381",1063 "repo": "core",1064 "path": "tests/components/ialarm_xr/test_init.py",1065 "file_name": "test_init.py",1066 "fun_name": "test_setup_not_ready",1067 "commit_message": "Create iAlarmXR integration (#67817)\n\n* Creating iAlarmXR integration\r\n\r\n* fixing after review code\r\n\r\n* fixing remaining review hints\r\n\r\n* fixing remaining review hints\r\n\r\n* updating underlying pyialarm library\r\n\r\n* Creating iAlarmXR integration\r\n\r\n* fixing after review code\r\n\r\n* fixing remaining review hints\r\n\r\n* fixing remaining review hints\r\n\r\n* updating underlying pyialarm library\r\n\r\n* fixing after iMicknl review\r\n\r\n* Improving exception handling\r\n\r\n* Updating pyialarmxr library\r\n\r\n* fixing after merge dev\r\n\r\n* fixing after iMicknl review\r\n\r\n* Update CODEOWNERS\r\n\r\nCo-authored-by: Ludovico de Nittis <git@denittis.one>\r\n\r\n* fixing iot_class\r\n\r\n* Update homeassistant/components/ialarmxr/config_flow.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* fixing after bdraco review\r\n\r\n* Update homeassistant/components/ialarmxr/config_flow.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* reverting catching exception in setup step\r\n\r\n* Update homeassistant/components/ialarmxr/__init__.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/__init__.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* fixing after bdraco suggestions\r\n\r\n* Update homeassistant/components/ialarmxr/alarm_control_panel.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/alarm_control_panel.py\r\n\r\nCo-authored-by: Mick Vleeshouwer <mick@imick.nl>\r\n\r\n* Update homeassistant/components/ialarmxr/config_flow.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/config_flow.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/__init__.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/__init__.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* Update homeassistant/components/ialarmxr/utils.py\r\n\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\n\r\n* regenerate translation and rename function to async_get_ialarmxr_mac\r\n\r\n* removing and collapsing unused error messages\r\n\r\n* fixing tests\r\n\r\n* improve code coverage in tests\r\n\r\n* improve code coverage in tests\r\n\r\n* improve code coverage in tests\r\n\r\n* fixing retry policy with new pyalarmxr library\r\n\r\n* snake case fix\r\n\r\n* renaming integration in ialarm_xr\r\n\r\n* renaming control panel name\r\n\r\nCo-authored-by: Ludovico de Nittis <git@denittis.one>\r\nCo-authored-by: J. Nick Koston <nick@koston.org>\r\nCo-authored-by: Mick Vleeshouwer <mick@imick.nl>",1068 "code": "async def test_setup_not_ready(hass, ialarmxr_api, mock_config_entry):\n \n ialarmxr_api.return_value.get_mac = Mock(side_effect=ConnectionError)\n\n mock_config_entry.add_to_hass(hass)\n assert not await hass.config_entries.async_setup(mock_config_entry.entry_id)\n await hass.async_block_till_done()\n assert mock_config_entry.state is ConfigEntryState.SETUP_RETRY\n\n",1069 "url": "https://github.com/home-assistant/core.git",1070 "language": "Python",1071 "ast_errors": "",1072 "n_ast_errors": 0,1073 "ast_levels": 10,1074 "n_whitespaces": 37,1075 "n_words": 19,1076 "vocab_size": 17,1077 "complexity": 1,1078 "nloc": 6,1079 "token_counts": 55,1080 "n_ast_nodes": 91,1081 "n_identifiers": 171082 },1083 {1084 "id": 264031,1085 "commit_id": "d789a7daa7712716c89259b987349917a89aece7",1086 "repo": "pyinstaller",1087 "path": "PyInstaller/utils/hooks/qt/__init__.py",1088 "file_name": "__init__.py",1089 "fun_name": "collect_qtqml_files",1090 "commit_message": "hookutils: reorganize the Qt hook utilities\n\nReorganize the Qt module information to provide information necessary\nto deal with variations between different python Qt bindings (PySide2,\nPyQt5, PySide6, and PyQt6). Replace the existing table-like dictionary\nwith list of entries, which is easier to format and document. From this\nlist, we now generate two dictionaries; one that maps Qt module (shared\nlibrary) names to the module info entries (the same role as the old\ndictionary), and one that maps python module names to the module info\nentries. The latter is necessary to accommodate python modules that do\nnot have corresponding Qt shared libraries (header-only Qt modules,\nsuch as QtAxContainer; or statically-linked module, such as QSci), but\nwe still need to provide information about plugins or translation\nfiles.\n\nThe new information list is based on manual inspection of source code\nfor Qt 5.15 and 6.3, and should provide comprehensive information about\nall plugin names and translation file basenames.\n\nIn addition, most of the helper functions, which take a reference to\nthe `QtLibraryInfo` class as their first argument, have been turned\ninto methods of the `QtLibraryInfo` class. The corresponding hooks\nhave also been adjusted.",1091 "code": "def collect_qtqml_files(self):\n \n\n # No-op if requested Qt-based package is not available.\n if self.version is None:\n return [], []\n\n # Not all PyQt5/PySide2 installs have QML files. In this case, location['Qml2ImportsPath'] is empty.\n # Furthermore, even if location path is provided, the directory itself may not exist.\n #\n # https://github.com/pyinstaller/pyinstaller/pull/3229#issuecomment-359735031\n # https://github.com/pyinstaller/pyinstaller/issues/3864\n #\n # In Qt 6, Qml2ImportsPath was deprecated in favor of QmlImportsPath. The former is not available in PySide6\n # 6.4.0 anymore (but is in PyQt6 6.4.0). Use the new QmlImportsPath if available.\n if 'QmlImportsPath' in self.location:\n qml_src_dir = self.location['QmlImportsPath']\n else:\n qml_src_dir = self.location['Qml2ImportsPath']\n if not qml_src_dir or not os.path.isdir(qml_src_dir):\n logger.warning('%s: QML directory %r does not exist. QML files not packaged.', self, qml_src_dir)\n return [], []\n\n qml_dst_dir = os.path.join(self.qt_rel_dir, 'qml')\n datas = [(qml_src_dir, qml_dst_dir)]\n binaries = [\n # Produce ``/path/to/Qt/Qml/path_to_qml_binary/qml_binary, PyQt5/Qt/Qml/path_to_qml_binary``.\n (\n qml_plugin_file,\n os.path.join(qml_dst_dir, os.path.dirname(os.path.relpath(qml_plugin_file, qml_src_dir)))\n ) for qml_plugin_file in misc.dlls_in_subdirs(qml_src_dir)\n ]\n\n return binaries, datas\n",1092 "url": "https://github.com/pyinstaller/pyinstaller.git",1093 "language": "Python",1094 "ast_errors": "",1095 "n_ast_errors": 0,1096 "ast_levels": 15,1097 "n_whitespaces": 397,1098 "n_words": 146,1099 "vocab_size": 99,1100 "complexity": 6,1101 "nloc": 19,1102 "token_counts": 144,1103 "n_ast_nodes": 243,1104 "n_identifiers": 201105 },1106 {1107 "id": 82418,1108 "commit_id": "c1290c9ff89cb00caa5469129fd527e9d82cd820",1109 "repo": "django-cms",1110 "path": "cms/tests/test_permmod.py",1111 "file_name": "test_permmod.py",1112 "fun_name": "test_patricks_move",1113 "commit_message": "ci: Added codespell (#7355)\n\nCo-authored-by: Christian Clauss <cclauss@me.com>\r\n\r\n* ci: codespell config taken from #7292",1114 "code": "def test_patricks_move(self):\n \n self.assertEqual(self.pg.node.parent, self.pe.node)\n # perform moves under slave...\n self.move_page(self.pg, self.pc)\n self.reload_pages()\n # page is now under PC\n self.assertEqual(self.pg.node.parent, self.pc.node)\n self.assertEqual(self.pg.get_absolute_url(), self.pg.publisher_public.get_absolute_url())\n self.move_page(self.pe, self.pg)\n self.reload_pages()\n self.assertEqual(self.pe.node.parent, self.pg.node)\n self.ph = self.ph.reload()\n # check urls - they should stay be the same now after the move\n self.assertEqual(\n self.pg.publisher_public.get_absolute_url(),\n self.pg.get_absolute_url()\n )\n self.assertEqual(\n self.ph.publisher_public.get_absolute_url(),\n self.ph.get_absolute_url()\n )\n\n # check if urls are correct after move\n self.assertEqual(\n self.pg.publisher_public.get_absolute_url(),\n '%smaster/slave-home/pc/pg/' % self.get_pages_root()\n )\n self.assertEqual(\n self.ph.publisher_public.get_absolute_url(),\n '%smaster/slave-home/pc/pg/pe/ph/' % self.get_pages_root()\n )\n\n",1115 "url": "https://github.com/django-cms/django-cms.git",1116 "language": "Python",1117 "ast_errors": "",1118 "n_ast_errors": 0,1119 "ast_levels": 11,1120 "n_whitespaces": 314,1121 "n_words": 72,1122 "vocab_size": 50,1123 "complexity": 1,1124 "nloc": 26,1125 "token_counts": 215,1126 "n_ast_nodes": 356,1127 "n_identifiers": 151128 },1129 {1130 "id": 291315,1131 "commit_id": "003e4224c89a6da381960dc5347750d1521d85c9",1132 "repo": "core",1133 "path": "tests/components/text/test_init.py",1134 "file_name": "test_init.py",1135 "fun_name": "test_text_new_min_max_pattern",1136 "commit_message": "Add `text` platform (#79454)\n\nCo-authored-by: Franck Nijhof <frenck@frenck.nl>\r\nCo-authored-by: Franck Nijhof <git@frenck.dev>",1137 "code": "async def test_text_new_min_max_pattern(hass):\n \n text = MockTextEntity(native_min=-1, native_max=500, pattern=r\"[a-z]\")\n text.hass = hass\n\n assert text.capability_attributes == {\n ATTR_MIN: 0,\n ATTR_MAX: MAX_LENGTH_STATE_STATE,\n ATTR_MODE: TextMode.TEXT,\n ATTR_PATTERN: r\"[a-z]\",\n }\n\n",1138 "url": "https://github.com/home-assistant/core.git",1139 "language": "Python",1140 "ast_errors": "",1141 "n_ast_errors": 0,1142 "ast_levels": 10,1143 "n_whitespaces": 67,1144 "n_words": 24,1145 "vocab_size": 23,1146 "complexity": 1,1147 "nloc": 9,1148 "token_counts": 55,1149 "n_ast_nodes": 85,1150 "n_identifiers": 151151 },1152 {1153 "id": 260017,1154 "commit_id": "71028322e8964cf1f341a7b293abaefeb5275e12",1155 "repo": "scikit-learn",1156 "path": "examples/text/plot_document_classification_20newsgroups.py",1157 "file_name": "plot_document_classification_20newsgroups.py",1158 "fun_name": "load_dataset",1159 "commit_message": "DOC rework plot_document_classification_20newsgroups.py example (#22928)\n\n\r\n\r\nCo-authored-by: Jérémie du Boisberranger <34657725+jeremiedbb@users.noreply.github.com>\r\nCo-authored-by: Olivier Grisel <olivier.grisel@ensta.org>\r\nCo-authored-by: Julien Jerphanion <git@jjerphan.xyz>",1160 "code": "def load_dataset(verbose=False, remove=()):\n \n\n data_train = fetch_20newsgroups(\n subset=\"train\",\n categories=categories,\n shuffle=True,\n random_state=42,\n remove=remove,\n )\n\n data_test = fetch_20newsgroups(\n subset=\"test\",\n categories=categories,\n shuffle=True,\n random_state=42,\n remove=remove,\n )\n\n # order of labels in `target_names` can be different from `categories`\n target_names = data_train.target_names\n\n # split target in a training set and a test set\n y_train, y_test = data_train.target, data_test.target\n\n # Extracting features from the training data using a sparse vectorizer\n t0 = time()\n vectorizer = TfidfVectorizer(\n sublinear_tf=True, max_df=0.5, min_df=5, stop_words=\"english\"\n )\n X_train = vectorizer.fit_transform(data_train.data)\n duration_train = time() - t0\n\n # Extracting features from the test data using the same vectorizer\n t0 = time()\n X_test = vectorizer.transform(data_test.data)\n duration_test = time() - t0\n\n feature_names = vectorizer.get_feature_names_out()\n\n if verbose:\n\n # compute size of loaded data\n data_train_size_mb = size_mb(data_train.data)\n data_test_size_mb = size_mb(data_test.data)\n\n print(\n f\"{len(data_train.data)} documents - \"\n f\"{data_train_size_mb:.2f}MB (training set)\"\n )\n print(f\"{len(data_test.data)} documents - {data_test_size_mb:.2f}MB (test set)\")\n print(f\"{len(target_names)} categories\")\n print(\n f\"vectorize training done in {duration_train:.3f}s \"\n f\"at {data_train_size_mb / duration_train:.3f}MB/s\"\n )\n print(f\"n_samples: {X_train.shape[0]}, n_features: {X_train.shape[1]}\")\n print(\n f\"vectorize testing done in {duration_test:.3f}s \"\n f\"at {data_test_size_mb / duration_test:.3f}MB/s\"\n )\n print(f\"n_samples: {X_test.shape[0]}, n_features: {X_test.shape[1]}\")\n\n return X_train, X_test, y_train, y_test, feature_names, target_names\n\n\n# %%\n# Compare feature effects\n# -----------------------\n# We train a first classification model without attempting to strip the metadata\n# of the dataset.\n\nX_train, X_test, y_train, y_test, feature_names, target_names = load_dataset(\n verbose=True\n)\n\n# %%\n# Our first model is an instance of the\n# :class:`~sklearn.linear_model.RidgeClassifier` class. This is a linear\n# classification model that uses the mean squared error on {-1, 1} encoded\n# targets, one for each possible class. Contrary to\n# :class:`~sklearn.linear_model.LogisticRegression`,\n# :class:`~sklearn.linear_model.RidgeClassifier` does not\n# provide probabilistic predictions (no `predict_proba` method),\n# but it is often faster to train.\n\nfrom sklearn.linear_model import RidgeClassifier\n\nclf = RidgeClassifier(tol=1e-2, solver=\"sparse_cg\")\nclf.fit(X_train, y_train)\npred = clf.predict(X_test)\n\n# %%\n# We plot the confusion matrix of this classifier to find if there is a pattern\n# in the classification errors.\n\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nfig, ax = plt.subplots(figsize=(10, 5))\nConfusionMatrixDisplay.from_predictions(y_test, pred, ax=ax)\nax.xaxis.set_ticklabels(target_names)\nax.yaxis.set_ticklabels(target_names)\n_ = ax.set_title(\n f\"Confusion Matrix for {clf.__class__.__name__}\\non the original documents\"\n)\n\n# %%\n# The confusion matrix highlights that documents of the `alt.atheism` class are\n# often confused with documents with the class `talk.religion.misc` class and\n# vice-versa which is expected since the topics are semantically related.\n#\n# We also observe that some documents of the `sci.space` class can be misclassified as\n# `comp.graphics` while the converse is much rarer. A manual inspection of those\n# badly classified documents would be required to get some insights on this\n# asymmetry. It could be the case that the vocabulary of the space topic could\n# be more specific than the vocabulary for computer graphics.\n#\n# We can gain a deeper understanding of how this classifier makes its decisions\n# by looking at the words with the highest average feature effects:\n\nimport pandas as pd\nimport numpy as np\n\n",1161 "url": "https://github.com/scikit-learn/scikit-learn.git",1162 "language": "Python",1163 "ast_errors": "",1164 "n_ast_errors": 0,1165 "ast_levels": 15,1166 "n_whitespaces": 735,1167 "n_words": 475,1168 "vocab_size": 266,1169 "complexity": 2,1170 "nloc": 48,1171 "token_counts": 224,1172 "n_ast_nodes": 713,1173 "n_identifiers": 671174 },1175 {1176 "id": 155176,1177 "commit_id": "193505fdf0c984743397ba3df56262f30aee13a8",1178 "repo": "modin",1179 "path": "modin/core/execution/unidist/implementations/pandas_on_unidist/partitioning/partition.py",1180 "file_name": "partition.py",1181 "fun_name": "apply",1182 "commit_message": "FEAT-#5053: Add pandas on unidist execution with MPI backend (#5059)\n\nSigned-off-by: Igoshev, Iaroslav <iaroslav.igoshev@intel.com>",1183 "code": "def apply(self, func, *args, **kwargs):\n \n logger = get_logger()\n logger.debug(f\"ENTER::Partition.apply::{self._identity}\")\n data = self._data\n call_queue = self.call_queue + [[func, args, kwargs]]\n if len(call_queue) > 1:\n logger.debug(f\"SUBMIT::_apply_list_of_funcs::{self._identity}\")\n result, length, width, ip = _apply_list_of_funcs.remote(call_queue, data)\n else:\n # We handle `len(call_queue) == 1` in a different way because\n # this dramatically improves performance.\n result, length, width, ip = _apply_func.remote(data, func, *args, **kwargs)\n logger.debug(f\"SUBMIT::_apply_func::{self._identity}\")\n logger.debug(f\"EXIT::Partition.apply::{self._identity}\")\n return PandasOnUnidistDataframePartition(result, length, width, ip)\n",1184 "url": "https://github.com/modin-project/modin.git",1185 "language": "Python",1186 "ast_errors": "",1187 "n_ast_errors": 0,1188 "ast_levels": 13,1189 "n_whitespaces": 193,1190 "n_words": 64,1191 "vocab_size": 51,1192 "complexity": 2,1193 "nloc": 13,1194 "token_counts": 126,1195 "n_ast_nodes": 222,1196 "n_identifiers": 211197 },1198 {1199 "id": 157201,1200 "commit_id": "b1e468e8645baee30992fbfa84250d816ac1098a",