codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import email.feedparser4import email.header5import email.message6import email.parser7import email.policy8import keyword9import pathlib10import sys11import typing12from typing import (13 Any,14 Callable,15 Generic,16 Literal,17 TypedDict,18 cast,19)20 21from . import licenses, requirements, specifiers, utils22from . import version as version_module23 24if typing.TYPE_CHECKING:25 from .licenses import NormalizedLicenseExpression26 27T = typing.TypeVar("T")28 29 30if sys.version_info >= (3, 11): # pragma: no cover31 ExceptionGroup = ExceptionGroup # noqa: F82132else: # pragma: no cover33 34 class ExceptionGroup(Exception):35 """A minimal implementation of :external:exc:`ExceptionGroup` from Python 3.11.36 37 If :external:exc:`ExceptionGroup` is already defined by Python itself,38 that version is used instead.39 """40 41 message: str42 exceptions: list[Exception]43 44 def __init__(self, message: str, exceptions: list[Exception]) -> None:45 self.message = message46 self.exceptions = exceptions47 48 def __repr__(self) -> str:49 return f"{self.__class__.__name__}({self.message!r}, {self.exceptions!r})"50 51 52class InvalidMetadata(ValueError):53 """A metadata field contains invalid data."""54 55 field: str56 """The name of the field that contains invalid data."""57 58 def __init__(self, field: str, message: str) -> None:59 self.field = field60 super().__init__(message)61 62 63# The RawMetadata class attempts to make as few assumptions about the underlying64# serialization formats as possible. The idea is that as long as a serialization65# formats offer some very basic primitives in *some* way then we can support66# serializing to and from that format.67class RawMetadata(TypedDict, total=False):68 """A dictionary of raw core metadata.69 70 Each field in core metadata maps to a key of this dictionary (when data is71 provided). The key is lower-case and underscores are used instead of dashes72 compared to the equivalent core metadata field. Any core metadata field that73 can be specified multiple times or can hold multiple values in a single74 field have a key with a plural name. See :class:`Metadata` whose attributes75 match the keys of this dictionary.76 77 Core metadata fields that can be specified multiple times are stored as a78 list or dict depending on which is appropriate for the field. Any fields79 which hold multiple values in a single field are stored as a list.80 81 """82 83 # Metadata 1.0 - PEP 24184 metadata_version: str85 name: str86 version: str87 platforms: list[str]88 summary: str89 description: str90 keywords: list[str]91 home_page: str92 author: str93 author_email: str94 license: str95 96 # Metadata 1.1 - PEP 31497 supported_platforms: list[str]98 download_url: str99 classifiers: list[str]100 requires: list[str]101 provides: list[str]102 obsoletes: list[str]103 104 # Metadata 1.2 - PEP 345105 maintainer: str106 maintainer_email: str107 requires_dist: list[str]108 provides_dist: list[str]109 obsoletes_dist: list[str]110 requires_python: str111 requires_external: list[str]112 project_urls: dict[str, str]113 114 # Metadata 2.0115 # PEP 426 attempted to completely revamp the metadata format116 # but got stuck without ever being able to build consensus on117 # it and ultimately ended up withdrawn.118 #119 # However, a number of tools had started emitting METADATA with120 # `2.0` Metadata-Version, so for historical reasons, this version121 # was skipped.122 123 # Metadata 2.1 - PEP 566124 description_content_type: str125 provides_extra: list[str]126 127 # Metadata 2.2 - PEP 643128 dynamic: list[str]129 130 # Metadata 2.3 - PEP 685131 # No new fields were added in PEP 685, just some edge case were132 # tightened up to provide better interoperability.133 134 # Metadata 2.4 - PEP 639135 license_expression: str136 license_files: list[str]137 138 # Metadata 2.5 - PEP 794139 import_names: list[str]140 import_namespaces: list[str]141 142 143# 'keywords' is special as it's a string in the core metadata spec, but we144# represent it as a list.145_STRING_FIELDS = {146 "author",147 "author_email",148 "description",149 "description_content_type",150 "download_url",151 "home_page",152 "license",153 "license_expression",154 "maintainer",155 "maintainer_email",156 "metadata_version",157 "name",158 "requires_python",159 "summary",160 "version",161}162 163_LIST_FIELDS = {164 "classifiers",165 "dynamic",166 "license_files",167 "obsoletes",168 "obsoletes_dist",169 "platforms",170 "provides",171 "provides_dist",172 "provides_extra",173 "requires",174 "requires_dist",175 "requires_external",176 "supported_platforms",177 "import_names",178 "import_namespaces",179}180 181_DICT_FIELDS = {182 "project_urls",183}184 185 186def _parse_keywords(data: str) -> list[str]:187 """Split a string of comma-separated keywords into a list of keywords."""188 return [k.strip() for k in data.split(",")]189 190 191def _parse_project_urls(data: list[str]) -> dict[str, str]:192 """Parse a list of label/URL string pairings separated by a comma."""193 urls = {}194 for pair in data:195 # Our logic is slightly tricky here as we want to try and do196 # *something* reasonable with malformed data.197 #198 # The main thing that we have to worry about, is data that does199 # not have a ',' at all to split the label from the Value. There200 # isn't a singular right answer here, and we will fail validation201 # later on (if the caller is validating) so it doesn't *really*202 # matter, but since the missing value has to be an empty str203 # and our return value is dict[str, str], if we let the key204 # be the missing value, then they'd have multiple '' values that205 # overwrite each other in a accumulating dict.206 #207 # The other potential issue is that it's possible to have the208 # same label multiple times in the metadata, with no solid "right"209 # answer with what to do in that case. As such, we'll do the only210 # thing we can, which is treat the field as unparsable and add it211 # to our list of unparsed fields.212 #213 # TODO: The spec doesn't say anything about if the keys should be214 # considered case sensitive or not... logically they should215 # be case-preserving and case-insensitive, but doing that216 # would open up more cases where we might have duplicate217 # entries.218 label, _, url = (s.strip() for s in pair.partition(","))219 220 if label in urls:221 # The label already exists in our set of urls, so this field222 # is unparsable, and we can just add the whole thing to our223 # unparsable data and stop processing it.224 raise KeyError("duplicate labels in project urls")225 urls[label] = url226 227 return urls228 229 230def _get_payload(msg: email.message.Message, source: bytes | str) -> str:231 """Get the body of the message."""232 # If our source is a str, then our caller has managed encodings for us,233 # and we don't need to deal with it.234 if isinstance(source, str):235 payload = msg.get_payload()236 assert isinstance(payload, str)237 return payload238 # If our source is a bytes, then we're managing the encoding and we need239 # to deal with it.240 else:241 bpayload = msg.get_payload(decode=True)242 assert isinstance(bpayload, bytes)243 try:244 return bpayload.decode("utf8", "strict")245 except UnicodeDecodeError as exc:246 raise ValueError("payload in an invalid encoding") from exc247 248 249# The various parse_FORMAT functions here are intended to be as lenient as250# possible in their parsing, while still returning a correctly typed251# RawMetadata.252#253# To aid in this, we also generally want to do as little touching of the254# data as possible, except where there are possibly some historic holdovers255# that make valid data awkward to work with.256#257# While this is a lower level, intermediate format than our ``Metadata``258# class, some light touch ups can make a massive difference in usability.259 260# Map METADATA fields to RawMetadata.261_EMAIL_TO_RAW_MAPPING = {262 "author": "author",263 "author-email": "author_email",264 "classifier": "classifiers",265 "description": "description",266 "description-content-type": "description_content_type",267 "download-url": "download_url",268 "dynamic": "dynamic",269 "home-page": "home_page",270 "import-name": "import_names",271 "import-namespace": "import_namespaces",272 "keywords": "keywords",273 "license": "license",274 "license-expression": "license_expression",275 "license-file": "license_files",276 "maintainer": "maintainer",277 "maintainer-email": "maintainer_email",278 "metadata-version": "metadata_version",279 "name": "name",280 "obsoletes": "obsoletes",281 "obsoletes-dist": "obsoletes_dist",282 "platform": "platforms",283 "project-url": "project_urls",284 "provides": "provides",285 "provides-dist": "provides_dist",286 "provides-extra": "provides_extra",287 "requires": "requires",288 "requires-dist": "requires_dist",289 "requires-external": "requires_external",290 "requires-python": "requires_python",291 "summary": "summary",292 "supported-platform": "supported_platforms",293 "version": "version",294}295_RAW_TO_EMAIL_MAPPING = {raw: email for email, raw in _EMAIL_TO_RAW_MAPPING.items()}296 297 298# This class is for writing RFC822 messages299class RFC822Policy(email.policy.EmailPolicy):300 """301 This is :class:`email.policy.EmailPolicy`, but with a simple ``header_store_parse``302 implementation that handles multi-line values, and some nice defaults.303 """304 305 utf8 = True306 mangle_from_ = False307 max_line_length = 0308 309 def header_store_parse(self, name: str, value: str) -> tuple[str, str]:310 size = len(name) + 2311 value = value.replace("\n", "\n" + " " * size)312 return (name, value)313 314 315# This class is for writing RFC822 messages316class RFC822Message(email.message.EmailMessage):317 """318 This is :class:`email.message.EmailMessage` with two small changes: it defaults to319 our `RFC822Policy`, and it correctly writes unicode when being called320 with `bytes()`.321 """322 323 def __init__(self) -> None:324 super().__init__(policy=RFC822Policy())325 326 def as_bytes(327 self, unixfrom: bool = False, policy: email.policy.Policy | None = None328 ) -> bytes:329 """330 Return the bytes representation of the message.331 332 This handles unicode encoding.333 """334 return self.as_string(unixfrom, policy=policy).encode("utf-8")335 336 337def parse_email(data: bytes | str) -> tuple[RawMetadata, dict[str, list[str]]]:338 """Parse a distribution's metadata stored as email headers (e.g. from ``METADATA``).339 340 This function returns a two-item tuple of dicts. The first dict is of341 recognized fields from the core metadata specification. Fields that can be342 parsed and translated into Python's built-in types are converted343 appropriately. All other fields are left as-is. Fields that are allowed to344 appear multiple times are stored as lists.345 346 The second dict contains all other fields from the metadata. This includes347 any unrecognized fields. It also includes any fields which are expected to348 be parsed into a built-in type but were not formatted appropriately. Finally,349 any fields that are expected to appear only once but are repeated are350 included in this dict.351 352 """353 raw: dict[str, str | list[str] | dict[str, str]] = {}354 unparsed: dict[str, list[str]] = {}355 356 if isinstance(data, str):357 parsed = email.parser.Parser(policy=email.policy.compat32).parsestr(data)358 else:359 parsed = email.parser.BytesParser(policy=email.policy.compat32).parsebytes(data)360 361 # We have to wrap parsed.keys() in a set, because in the case of multiple362 # values for a key (a list), the key will appear multiple times in the363 # list of keys, but we're avoiding that by using get_all().364 for name_with_case in frozenset(parsed.keys()):365 # Header names in RFC are case insensitive, so we'll normalize to all366 # lower case to make comparisons easier.367 name = name_with_case.lower()368 369 # We use get_all() here, even for fields that aren't multiple use,370 # because otherwise someone could have e.g. two Name fields, and we371 # would just silently ignore it rather than doing something about it.372 headers = parsed.get_all(name) or []373 374 # The way the email module works when parsing bytes is that it375 # unconditionally decodes the bytes as ascii using the surrogateescape376 # handler. When you pull that data back out (such as with get_all() ),377 # it looks to see if the str has any surrogate escapes, and if it does378 # it wraps it in a Header object instead of returning the string.379 #380 # As such, we'll look for those Header objects, and fix up the encoding.381 value = []382 # Flag if we have run into any issues processing the headers, thus383 # signalling that the data belongs in 'unparsed'.384 valid_encoding = True385 for h in headers:386 # It's unclear if this can return more types than just a Header or387 # a str, so we'll just assert here to make sure.388 assert isinstance(h, (email.header.Header, str))389 390 # If it's a header object, we need to do our little dance to get391 # the real data out of it. In cases where there is invalid data392 # we're going to end up with mojibake, but there's no obvious, good393 # way around that without reimplementing parts of the Header object394 # ourselves.395 #396 # That should be fine since, if mojibacked happens, this key is397 # going into the unparsed dict anyways.398 if isinstance(h, email.header.Header):399 # The Header object stores it's data as chunks, and each chunk400 # can be independently encoded, so we'll need to check each401 # of them.402 chunks: list[tuple[bytes, str | None]] = []403 for binary, _encoding in email.header.decode_header(h):404 try:405 binary.decode("utf8", "strict")406 except UnicodeDecodeError:407 # Enable mojibake.408 encoding = "latin1"409 valid_encoding = False410 else:411 encoding = "utf8"412 chunks.append((binary, encoding))413 414 # Turn our chunks back into a Header object, then let that415 # Header object do the right thing to turn them into a416 # string for us.417 value.append(str(email.header.make_header(chunks)))418 # This is already a string, so just add it.419 else:420 value.append(h)421 422 # We've processed all of our values to get them into a list of str,423 # but we may have mojibake data, in which case this is an unparsed424 # field.425 if not valid_encoding:426 unparsed[name] = value427 continue428 429 raw_name = _EMAIL_TO_RAW_MAPPING.get(name)430 if raw_name is None:431 # This is a bit of a weird situation, we've encountered a key that432 # we don't know what it means, so we don't know whether it's meant433 # to be a list or not.434 #435 # Since we can't really tell one way or another, we'll just leave it436 # as a list, even though it may be a single item list, because that's437 # what makes the most sense for email headers.438 unparsed[name] = value439 continue440 441 # If this is one of our string fields, then we'll check to see if our442 # value is a list of a single item. If it is then we'll assume that443 # it was emitted as a single string, and unwrap the str from inside444 # the list.445 #446 # If it's any other kind of data, then we haven't the faintest clue447 # what we should parse it as, and we have to just add it to our list448 # of unparsed stuff.449 if raw_name in _STRING_FIELDS and len(value) == 1:450 raw[raw_name] = value[0]451 # If this is import_names, we need to special case the empty field452 # case, which converts to an empty list instead of None. We can't let453 # the empty case slip through, as it will fail validation.454 elif raw_name == "import_names" and value == [""]:455 raw[raw_name] = []456 # If this is one of our list of string fields, then we can just assign457 # the value, since email *only* has strings, and our get_all() call458 # above ensures that this is a list.459 elif raw_name in _LIST_FIELDS:460 raw[raw_name] = value461 # Special Case: Keywords462 # The keywords field is implemented in the metadata spec as a str,463 # but it conceptually is a list of strings, and is serialized using464 # ", ".join(keywords), so we'll do some light data massaging to turn465 # this into what it logically is.466 elif raw_name == "keywords" and len(value) == 1:467 raw[raw_name] = _parse_keywords(value[0])468 # Special Case: Project-URL469 # The project urls is implemented in the metadata spec as a list of470 # specially-formatted strings that represent a key and a value, which471 # is fundamentally a mapping, however the email format doesn't support472 # mappings in a sane way, so it was crammed into a list of strings473 # instead.474 #475 # We will do a little light data massaging to turn this into a map as476 # it logically should be.477 elif raw_name == "project_urls":478 try:479 raw[raw_name] = _parse_project_urls(value)480 except KeyError:481 unparsed[name] = value482 # Nothing that we've done has managed to parse this, so it'll just483 # throw it in our unparsable data and move on.484 else:485 unparsed[name] = value486 487 # We need to support getting the Description from the message payload in488 # addition to getting it from the the headers. This does mean, though, there489 # is the possibility of it being set both ways, in which case we put both490 # in 'unparsed' since we don't know which is right.491 try:492 payload = _get_payload(parsed, data)493 except ValueError:494 unparsed.setdefault("description", []).append(495 parsed.get_payload(decode=isinstance(data, bytes)) # type: ignore[call-overload]496 )497 else:498 if payload:499 # Check to see if we've already got a description, if so then both500 # it, and this body move to unparsable.501 if "description" in raw:502 description_header = cast("str", raw.pop("description"))503 unparsed.setdefault("description", []).extend(504 [description_header, payload]505 )506 elif "description" in unparsed:507 unparsed["description"].append(payload)508 else:509 raw["description"] = payload510 511 # We need to cast our `raw` to a metadata, because a TypedDict only support512 # literal key names, but we're computing our key names on purpose, but the513 # way this function is implemented, our `TypedDict` can only have valid key514 # names.515 return cast("RawMetadata", raw), unparsed516 517 518_NOT_FOUND = object()519 520 521# Keep the two values in sync.522_VALID_METADATA_VERSIONS = ["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]523_MetadataVersion = Literal["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]524 525_REQUIRED_ATTRS = frozenset(["metadata_version", "name", "version"])526 527 528class _Validator(Generic[T]):529 """Validate a metadata field.530 531 All _process_*() methods correspond to a core metadata field. The method is532 called with the field's raw value. If the raw value is valid it is returned533 in its "enriched" form (e.g. ``version.Version`` for the ``Version`` field).534 If the raw value is invalid, :exc:`InvalidMetadata` is raised (with a cause535 as appropriate).536 """537 538 name: str539 raw_name: str540 added: _MetadataVersion541 542 def __init__(543 self,544 *,545 added: _MetadataVersion = "1.0",546 ) -> None:547 self.added = added548 549 def __set_name__(self, _owner: Metadata, name: str) -> None:550 self.name = name551 self.raw_name = _RAW_TO_EMAIL_MAPPING[name]552 553 def __get__(self, instance: Metadata, _owner: type[Metadata]) -> T:554 # With Python 3.8, the caching can be replaced with functools.cached_property().555 # No need to check the cache as attribute lookup will resolve into the556 # instance's __dict__ before __get__ is called.557 cache = instance.__dict__558 value = instance._raw.get(self.name)559 560 # To make the _process_* methods easier, we'll check if the value is None561 # and if this field is NOT a required attribute, and if both of those562 # things are true, we'll skip the the converter. This will mean that the563 # converters never have to deal with the None union.564 if self.name in _REQUIRED_ATTRS or value is not None:565 try:566 converter: Callable[[Any], T] = getattr(self, f"_process_{self.name}")567 except AttributeError:568 pass569 else:570 value = converter(value)571 572 cache[self.name] = value573 try:574 del instance._raw[self.name] # type: ignore[misc]575 except KeyError:576 pass577 578 return cast("T", value)579 580 def _invalid_metadata(581 self, msg: str, cause: Exception | None = None582 ) -> InvalidMetadata:583 exc = InvalidMetadata(584 self.raw_name, msg.format_map({"field": repr(self.raw_name)})585 )586 exc.__cause__ = cause587 return exc588 589 def _process_metadata_version(self, value: str) -> _MetadataVersion:590 # Implicitly makes Metadata-Version required.591 if value not in _VALID_METADATA_VERSIONS:592 raise self._invalid_metadata(f"{value!r} is not a valid metadata version")593 return cast("_MetadataVersion", value)594 595 def _process_name(self, value: str) -> str:596 if not value:597 raise self._invalid_metadata("{field} is a required field")598 # Validate the name as a side-effect.599 try:600 utils.canonicalize_name(value, validate=True)601 except utils.InvalidName as exc:602 raise self._invalid_metadata(603 f"{value!r} is invalid for {{field}}", cause=exc604 ) from exc605 else:606 return value607 608 def _process_version(self, value: str) -> version_module.Version:609 if not value:610 raise self._invalid_metadata("{field} is a required field")611 try:612 return version_module.parse(value)613 except version_module.InvalidVersion as exc:614 raise self._invalid_metadata(615 f"{value!r} is invalid for {{field}}", cause=exc616 ) from exc617 618 def _process_summary(self, value: str) -> str:619 """Check the field contains no newlines."""620 if "\n" in value:621 raise self._invalid_metadata("{field} must be a single line")622 return value623 624 def _process_description_content_type(self, value: str) -> str:625 content_types = {"text/plain", "text/x-rst", "text/markdown"}626 message = email.message.EmailMessage()627 message["content-type"] = value628 629 content_type, parameters = (630 # Defaults to `text/plain` if parsing failed.631 message.get_content_type().lower(),632 message["content-type"].params,633 )634 # Check if content-type is valid or defaulted to `text/plain` and thus was635 # not parseable.636 if content_type not in content_types or content_type not in value.lower():637 raise self._invalid_metadata(638 f"{{field}} must be one of {list(content_types)}, not {value!r}"639 )640 641 charset = parameters.get("charset", "UTF-8")642 if charset != "UTF-8":643 raise self._invalid_metadata(644 f"{{field}} can only specify the UTF-8 charset, not {list(charset)}"645 )646 647 markdown_variants = {"GFM", "CommonMark"}648 variant = parameters.get("variant", "GFM") # Use an acceptable default.649 if content_type == "text/markdown" and variant not in markdown_variants:650 raise self._invalid_metadata(651 f"valid Markdown variants for {{field}} are {list(markdown_variants)}, "652 f"not {variant!r}",653 )654 return value655 656 def _process_dynamic(self, value: list[str]) -> list[str]:657 for dynamic_field in map(str.lower, value):658 if dynamic_field in {"name", "version", "metadata-version"}:659 raise self._invalid_metadata(660 f"{dynamic_field!r} is not allowed as a dynamic field"661 )662 elif dynamic_field not in _EMAIL_TO_RAW_MAPPING:663 raise self._invalid_metadata(664 f"{dynamic_field!r} is not a valid dynamic field"665 )666 return list(map(str.lower, value))667 668 def _process_provides_extra(669 self,670 value: list[str],671 ) -> list[utils.NormalizedName]:672 normalized_names = []673 try:674 for name in value:675 normalized_names.append(utils.canonicalize_name(name, validate=True))676 except utils.InvalidName as exc:677 raise self._invalid_metadata(678 f"{name!r} is invalid for {{field}}", cause=exc679 ) from exc680 else:681 return normalized_names682 683 def _process_requires_python(self, value: str) -> specifiers.SpecifierSet:684 try:685 return specifiers.SpecifierSet(value)686 except specifiers.InvalidSpecifier as exc:687 raise self._invalid_metadata(688 f"{value!r} is invalid for {{field}}", cause=exc689 ) from exc690 691 def _process_requires_dist(692 self,693 value: list[str],694 ) -> list[requirements.Requirement]:695 reqs = []696 try:697 for req in value:698 reqs.append(requirements.Requirement(req))699 except requirements.InvalidRequirement as exc:700 raise self._invalid_metadata(701 f"{req!r} is invalid for {{field}}", cause=exc702 ) from exc703 else:704 return reqs705 706 def _process_license_expression(self, value: str) -> NormalizedLicenseExpression:707 try:708 return licenses.canonicalize_license_expression(value)709 except ValueError as exc:710 raise self._invalid_metadata(711 f"{value!r} is invalid for {{field}}", cause=exc712 ) from exc713 714 def _process_license_files(self, value: list[str]) -> list[str]:715 paths = []716 for path in value:717 if ".." in path:718 raise self._invalid_metadata(719 f"{path!r} is invalid for {{field}}, "720 "parent directory indicators are not allowed"721 )722 if "*" in path:723 raise self._invalid_metadata(724 f"{path!r} is invalid for {{field}}, paths must be resolved"725 )726 if (727 pathlib.PurePosixPath(path).is_absolute()728 or pathlib.PureWindowsPath(path).is_absolute()729 ):730 raise self._invalid_metadata(731 f"{path!r} is invalid for {{field}}, paths must be relative"732 )733 if pathlib.PureWindowsPath(path).as_posix() != path:734 raise self._invalid_metadata(735 f"{path!r} is invalid for {{field}}, paths must use '/' delimiter"736 )737 paths.append(path)738 return paths739 740 def _process_import_names(self, value: list[str]) -> list[str]:741 for import_name in value:742 name, semicolon, private = import_name.partition(";")743 name = name.rstrip()744 for identifier in name.split("."):745 if not identifier.isidentifier():746 raise self._invalid_metadata(747 f"{name!r} is invalid for {{field}}; "748 f"{identifier!r} is not a valid identifier"749 )750 elif keyword.iskeyword(identifier):751 raise self._invalid_metadata(752 f"{name!r} is invalid for {{field}}; "753 f"{identifier!r} is a keyword"754 )755 if semicolon and private.lstrip() != "private":756 raise self._invalid_metadata(757 f"{import_name!r} is invalid for {{field}}; "758 "the only valid option is 'private'"759 )760 return value761 762 _process_import_namespaces = _process_import_names763 764 765class Metadata:766 """Representation of distribution metadata.767 768 Compared to :class:`RawMetadata`, this class provides objects representing769 metadata fields instead of only using built-in types. Any invalid metadata770 will cause :exc:`InvalidMetadata` to be raised (with a771 :py:attr:`~BaseException.__cause__` attribute as appropriate).772 """773 774 _raw: RawMetadata775 776 @classmethod777 def from_raw(cls, data: RawMetadata, *, validate: bool = True) -> Metadata:778 """Create an instance from :class:`RawMetadata`.779 780 If *validate* is true, all metadata will be validated. All exceptions781 related to validation will be gathered and raised as an :class:`ExceptionGroup`.782 """783 ins = cls()784 ins._raw = data.copy() # Mutations occur due to caching enriched values.785 786 if validate:787 exceptions: list[Exception] = []788 try:789 metadata_version = ins.metadata_version790 metadata_age = _VALID_METADATA_VERSIONS.index(metadata_version)791 except InvalidMetadata as metadata_version_exc:792 exceptions.append(metadata_version_exc)793 metadata_version = None794 795 # Make sure to check for the fields that are present, the required796 # fields (so their absence can be reported).797 fields_to_check = frozenset(ins._raw) | _REQUIRED_ATTRS798 # Remove fields that have already been checked.799 fields_to_check -= {"metadata_version"}800 801 for key in fields_to_check:802 try:803 if metadata_version:804 # Can't use getattr() as that triggers descriptor protocol which805 # will fail due to no value for the instance argument.806 try:807 field_metadata_version = cls.__dict__[key].added808 except KeyError:809 exc = InvalidMetadata(key, f"unrecognized field: {key!r}")810 exceptions.append(exc)811 continue812 field_age = _VALID_METADATA_VERSIONS.index(813 field_metadata_version814 )815 if field_age > metadata_age:816 field = _RAW_TO_EMAIL_MAPPING[key]817 exc = InvalidMetadata(818 field,819 f"{field} introduced in metadata version "820 f"{field_metadata_version}, not {metadata_version}",821 )822 exceptions.append(exc)823 continue824 getattr(ins, key)825 except InvalidMetadata as exc:826 exceptions.append(exc)827 828 if exceptions:829 raise ExceptionGroup("invalid metadata", exceptions)830 831 return ins832 833 @classmethod834 def from_email(cls, data: bytes | str, *, validate: bool = True) -> Metadata:835 """Parse metadata from email headers.836 837 If *validate* is true, the metadata will be validated. All exceptions838 related to validation will be gathered and raised as an :class:`ExceptionGroup`.839 """840 raw, unparsed = parse_email(data)841 842 if validate:843 exceptions: list[Exception] = []844 for unparsed_key in unparsed:845 if unparsed_key in _EMAIL_TO_RAW_MAPPING:846 message = f"{unparsed_key!r} has invalid data"847 else:848 message = f"unrecognized field: {unparsed_key!r}"849 exceptions.append(InvalidMetadata(unparsed_key, message))850 851 if exceptions:852 raise ExceptionGroup("unparsed", exceptions)853 854 try:855 return cls.from_raw(raw, validate=validate)856 except ExceptionGroup as exc_group:857 raise ExceptionGroup(858 "invalid or unparsed metadata", exc_group.exceptions859 ) from None860 861 metadata_version: _Validator[_MetadataVersion] = _Validator()862 """:external:ref:`core-metadata-metadata-version`863 (required; validated to be a valid metadata version)"""864 # `name` is not normalized/typed to NormalizedName so as to provide access to865 # the original/raw name.866 name: _Validator[str] = _Validator()867 """:external:ref:`core-metadata-name`868 (required; validated using :func:`~packaging.utils.canonicalize_name` and its869 *validate* parameter)"""870 version: _Validator[version_module.Version] = _Validator()871 """:external:ref:`core-metadata-version` (required)"""872 dynamic: _Validator[list[str] | None] = _Validator(873 added="2.2",874 )875 """:external:ref:`core-metadata-dynamic`876 (validated against core metadata field names and lowercased)"""877 platforms: _Validator[list[str] | None] = _Validator()878 """:external:ref:`core-metadata-platform`"""879 supported_platforms: _Validator[list[str] | None] = _Validator(added="1.1")880 """:external:ref:`core-metadata-supported-platform`"""881 summary: _Validator[str | None] = _Validator()882 """:external:ref:`core-metadata-summary` (validated to contain no newlines)"""883 description: _Validator[str | None] = _Validator() # TODO 2.1: can be in body884 """:external:ref:`core-metadata-description`"""885 description_content_type: _Validator[str | None] = _Validator(added="2.1")886 """:external:ref:`core-metadata-description-content-type` (validated)"""887 keywords: _Validator[list[str] | None] = _Validator()888 """:external:ref:`core-metadata-keywords`"""889 home_page: _Validator[str | None] = _Validator()890 """:external:ref:`core-metadata-home-page`"""891 download_url: _Validator[str | None] = _Validator(added="1.1")892 """:external:ref:`core-metadata-download-url`"""893 author: _Validator[str | None] = _Validator()894 """:external:ref:`core-metadata-author`"""895 author_email: _Validator[str | None] = _Validator()896 """:external:ref:`core-metadata-author-email`"""897 maintainer: _Validator[str | None] = _Validator(added="1.2")898 """:external:ref:`core-metadata-maintainer`"""899 maintainer_email: _Validator[str | None] = _Validator(added="1.2")900 """:external:ref:`core-metadata-maintainer-email`"""901 license: _Validator[str | None] = _Validator()902 """:external:ref:`core-metadata-license`"""903 license_expression: _Validator[NormalizedLicenseExpression | None] = _Validator(904 added="2.4"905 )906 """:external:ref:`core-metadata-license-expression`"""907 license_files: _Validator[list[str] | None] = _Validator(added="2.4")908 """:external:ref:`core-metadata-license-file`"""909 classifiers: _Validator[list[str] | None] = _Validator(added="1.1")910 """:external:ref:`core-metadata-classifier`"""911 requires_dist: _Validator[list[requirements.Requirement] | None] = _Validator(912 added="1.2"913 )914 """:external:ref:`core-metadata-requires-dist`"""915 requires_python: _Validator[specifiers.SpecifierSet | None] = _Validator(916 added="1.2"917 )918 """:external:ref:`core-metadata-requires-python`"""919 # Because `Requires-External` allows for non-PEP 440 version specifiers, we920 # don't do any processing on the values.921 requires_external: _Validator[list[str] | None] = _Validator(added="1.2")922 """:external:ref:`core-metadata-requires-external`"""923 project_urls: _Validator[dict[str, str] | None] = _Validator(added="1.2")924 """:external:ref:`core-metadata-project-url`"""925 # PEP 685 lets us raise an error if an extra doesn't pass `Name` validation926 # regardless of metadata version.927 provides_extra: _Validator[list[utils.NormalizedName] | None] = _Validator(928 added="2.1",929 )930 """:external:ref:`core-metadata-provides-extra`"""931 provides_dist: _Validator[list[str] | None] = _Validator(added="1.2")932 """:external:ref:`core-metadata-provides-dist`"""933 obsoletes_dist: _Validator[list[str] | None] = _Validator(added="1.2")934 """:external:ref:`core-metadata-obsoletes-dist`"""935 import_names: _Validator[list[str] | None] = _Validator(added="2.5")936 """:external:ref:`core-metadata-import-name`"""937 import_namespaces: _Validator[list[str] | None] = _Validator(added="2.5")938 """:external:ref:`core-metadata-import-namespace`"""939 requires: _Validator[list[str] | None] = _Validator(added="1.1")940 """``Requires`` (deprecated)"""941 provides: _Validator[list[str] | None] = _Validator(added="1.1")942 """``Provides`` (deprecated)"""943 obsoletes: _Validator[list[str] | None] = _Validator(added="1.1")944 """``Obsoletes`` (deprecated)"""945 946 def as_rfc822(self) -> RFC822Message:947 """948 Return an RFC822 message with the metadata.949 """950 message = RFC822Message()951 self._write_metadata(message)952 return message953 954 def _write_metadata(self, message: RFC822Message) -> None:955 """956 Return an RFC822 message with the metadata.957 """958 for name, validator in self.__class__.__dict__.items():959 if isinstance(validator, _Validator) and name != "description":960 value = getattr(self, name)961 email_name = _RAW_TO_EMAIL_MAPPING[name]962 if value is not None:963 if email_name == "project-url":964 for label, url in value.items():965 message[email_name] = f"{label}, {url}"966 elif email_name == "keywords":967 message[email_name] = ",".join(value)968 elif email_name == "import-name" and value == []:969 message[email_name] = ""970 elif isinstance(value, list):971 for item in value:972 message[email_name] = str(item)973 else:974 message[email_name] = str(value)975 976 # The description is a special case because it is in the body of the message.977 if self.description is not None:978 message.set_payload(self.description)979 