codekingpro/portable-devtools
114k
1from __future__ import annotations2 3import email.header4import email.message5import email.parser6import email.policy7import keyword8import pathlib9import typing10from typing import (11 Any,12 Callable,13 Generic,14 Literal,15 TypedDict,16 cast,17)18 19from . import licenses, requirements, specifiers, utils20from . import version as version_module21from .errors import ExceptionGroup, _ErrorCollector22 23if typing.TYPE_CHECKING:24 from .licenses import NormalizedLicenseExpression25 26T = typing.TypeVar("T")27 28 29__all__ = [30 "ExceptionGroup", # Keep this for a bit (makes mypy happy w/ 26.0 compat)31 "InvalidMetadata",32 "Metadata",33 "RFC822Message",34 "RFC822Policy",35 "RawMetadata",36 "parse_email",37]38 39 40def __dir__() -> list[str]:41 return __all__42 43 44class InvalidMetadata(ValueError):45 """A metadata field contains invalid data."""46 47 field: str48 """The name of the field that contains invalid data."""49 50 def __init__(self, field: str, message: str) -> None:51 self.field = field52 super().__init__(message)53 54 55# The RawMetadata class attempts to make as few assumptions about the underlying56# serialization formats as possible. The idea is that as long as a serialization57# formats offer some very basic primitives in *some* way then we can support58# serializing to and from that format.59class RawMetadata(TypedDict, total=False):60 """A dictionary of raw core metadata.61 62 Each field in core metadata maps to a key of this dictionary (when data is63 provided). The key is lower-case and underscores are used instead of dashes64 compared to the equivalent core metadata field. Any core metadata field that65 can be specified multiple times or can hold multiple values in a single66 field have a key with a plural name. See :class:`Metadata` whose attributes67 match the keys of this dictionary.68 69 Core metadata fields that can be specified multiple times are stored as a70 list or dict depending on which is appropriate for the field. Any fields71 which hold multiple values in a single field are stored as a list. All fields72 are considered optional.73 """74 75 # Metadata 1.0 - PEP 24176 metadata_version: str77 name: str78 version: str79 platforms: list[str]80 summary: str81 description: str82 keywords: list[str]83 home_page: str84 author: str85 author_email: str86 license: str87 88 # Metadata 1.1 - PEP 31489 supported_platforms: list[str]90 download_url: str91 classifiers: list[str]92 requires: list[str]93 provides: list[str]94 obsoletes: list[str]95 96 # Metadata 1.2 - PEP 34597 maintainer: str98 maintainer_email: str99 requires_dist: list[str]100 provides_dist: list[str]101 obsoletes_dist: list[str]102 requires_python: str103 requires_external: list[str]104 project_urls: dict[str, str]105 106 # Metadata 2.0107 # PEP 426 attempted to completely revamp the metadata format108 # but got stuck without ever being able to build consensus on109 # it and ultimately ended up withdrawn.110 #111 # However, a number of tools had started emitting METADATA with112 # `2.0` Metadata-Version, so for historical reasons, this version113 # was skipped.114 115 # Metadata 2.1 - PEP 566116 description_content_type: str117 provides_extra: list[str]118 119 # Metadata 2.2 - PEP 643120 dynamic: list[str]121 122 # Metadata 2.3 - PEP 685123 # No new fields were added in PEP 685, just some edge case were124 # tightened up to provide better interoperability.125 126 # Metadata 2.4 - PEP 639127 license_expression: str128 license_files: list[str]129 130 # Metadata 2.5 - PEP 794131 import_names: list[str]132 import_namespaces: list[str]133 134 135# 'keywords' is special as it's a string in the core metadata spec, but we136# represent it as a list.137_STRING_FIELDS = {138 "author",139 "author_email",140 "description",141 "description_content_type",142 "download_url",143 "home_page",144 "license",145 "license_expression",146 "maintainer",147 "maintainer_email",148 "metadata_version",149 "name",150 "requires_python",151 "summary",152 "version",153}154 155_LIST_FIELDS = {156 "classifiers",157 "dynamic",158 "license_files",159 "obsoletes",160 "obsoletes_dist",161 "platforms",162 "provides",163 "provides_dist",164 "provides_extra",165 "requires",166 "requires_dist",167 "requires_external",168 "supported_platforms",169 "import_names",170 "import_namespaces",171}172 173_DICT_FIELDS = {174 "project_urls",175}176 177 178def _parse_keywords(data: str) -> list[str]:179 """Split a string of comma-separated keywords into a list of keywords."""180 return [k.strip() for k in data.split(",")]181 182 183def _parse_project_urls(data: list[str]) -> dict[str, str]:184 """Parse a list of label/URL string pairings separated by a comma."""185 urls = {}186 for pair in data:187 # Our logic is slightly tricky here as we want to try and do188 # *something* reasonable with malformed data.189 #190 # The main thing that we have to worry about, is data that does191 # not have a ',' at all to split the label from the Value. There192 # isn't a singular right answer here, and we will fail validation193 # later on (if the caller is validating) so it doesn't *really*194 # matter, but since the missing value has to be an empty str195 # and our return value is dict[str, str], if we let the key196 # be the missing value, then they'd have multiple '' values that197 # overwrite each other in a accumulating dict.198 #199 # The other potential issue is that it's possible to have the200 # same label multiple times in the metadata, with no solid "right"201 # answer with what to do in that case. As such, we'll do the only202 # thing we can, which is treat the field as unparsable and add it203 # to our list of unparsed fields.204 #205 # TODO: The spec doesn't say anything about if the keys should be206 # considered case sensitive or not... logically they should207 # be case-preserving and case-insensitive, but doing that208 # would open up more cases where we might have duplicate209 # entries.210 label, _, url = (s.strip() for s in pair.partition(","))211 212 if label in urls:213 # The label already exists in our set of urls, so this field214 # is unparsable, and we can just add the whole thing to our215 # unparsable data and stop processing it.216 raise KeyError("duplicate labels in project urls")217 urls[label] = url218 219 return urls220 221 222def _get_payload(msg: email.message.Message, source: bytes | str) -> str:223 """Get the body of the message."""224 # If our source is a str, then our caller has managed encodings for us,225 # and we don't need to deal with it.226 if isinstance(source, str):227 payload = msg.get_payload()228 assert isinstance(payload, str)229 return payload230 # If our source is a bytes, then we're managing the encoding and we need231 # to deal with it.232 else:233 bpayload = msg.get_payload(decode=True)234 assert isinstance(bpayload, bytes)235 try:236 return bpayload.decode("utf8", "strict")237 except UnicodeDecodeError as exc:238 raise ValueError("payload in an invalid encoding") from exc239 240 241# The various parse_FORMAT functions here are intended to be as lenient as242# possible in their parsing, while still returning a correctly typed243# RawMetadata.244#245# To aid in this, we also generally want to do as little touching of the246# data as possible, except where there are possibly some historic holdovers247# that make valid data awkward to work with.248#249# While this is a lower level, intermediate format than our ``Metadata``250# class, some light touch ups can make a massive difference in usability.251 252# Map METADATA fields to RawMetadata.253_EMAIL_TO_RAW_MAPPING = {254 "author": "author",255 "author-email": "author_email",256 "classifier": "classifiers",257 "description": "description",258 "description-content-type": "description_content_type",259 "download-url": "download_url",260 "dynamic": "dynamic",261 "home-page": "home_page",262 "import-name": "import_names",263 "import-namespace": "import_namespaces",264 "keywords": "keywords",265 "license": "license",266 "license-expression": "license_expression",267 "license-file": "license_files",268 "maintainer": "maintainer",269 "maintainer-email": "maintainer_email",270 "metadata-version": "metadata_version",271 "name": "name",272 "obsoletes": "obsoletes",273 "obsoletes-dist": "obsoletes_dist",274 "platform": "platforms",275 "project-url": "project_urls",276 "provides": "provides",277 "provides-dist": "provides_dist",278 "provides-extra": "provides_extra",279 "requires": "requires",280 "requires-dist": "requires_dist",281 "requires-external": "requires_external",282 "requires-python": "requires_python",283 "summary": "summary",284 "supported-platform": "supported_platforms",285 "version": "version",286}287_RAW_TO_EMAIL_MAPPING = {raw: email for email, raw in _EMAIL_TO_RAW_MAPPING.items()}288 289 290# This class is for writing RFC822 messages291class RFC822Policy(email.policy.EmailPolicy):292 """293 This is :class:`email.policy.EmailPolicy`, but with a simple ``header_store_parse``294 implementation that handles multi-line values, and some nice defaults.295 """296 297 utf8 = True298 mangle_from_ = False299 max_line_length = 0300 301 def header_store_parse(self, name: str, value: str) -> tuple[str, str]:302 size = len(name) + 2303 value = value.replace("\n", "\n" + " " * size)304 return (name, value)305 306 307# This class is for writing RFC822 messages308class RFC822Message(email.message.EmailMessage):309 """310 This is :class:`email.message.EmailMessage` with two small changes: it defaults to311 our `RFC822Policy`, and it correctly writes unicode when being called312 with `bytes()`.313 """314 315 def __init__(self) -> None:316 super().__init__(policy=RFC822Policy())317 318 def as_bytes(319 self, unixfrom: bool = False, policy: email.policy.Policy | None = None320 ) -> bytes:321 """322 Return the bytes representation of the message.323 324 This handles unicode encoding.325 """326 return self.as_string(unixfrom, policy=policy).encode("utf-8")327 328 329def parse_email(data: bytes | str) -> tuple[RawMetadata, dict[str, list[str]]]:330 """Parse a distribution's metadata stored as email headers (e.g. from ``METADATA``).331 332 This function returns a two-item tuple of dicts. The first dict is of333 recognized fields from the core metadata specification. Fields that can be334 parsed and translated into Python's built-in types are converted335 appropriately. All other fields are left as-is. Fields that are allowed to336 appear multiple times are stored as lists.337 338 The second dict contains all other fields from the metadata. This includes339 any unrecognized fields. It also includes any fields which are expected to340 be parsed into a built-in type but were not formatted appropriately. Finally,341 any fields that are expected to appear only once but are repeated are342 included in this dict.343 344 """345 raw: dict[str, str | list[str] | dict[str, str]] = {}346 unparsed: dict[str, list[str]] = {}347 348 if isinstance(data, str):349 parsed = email.parser.Parser(policy=email.policy.compat32).parsestr(data)350 else:351 parsed = email.parser.BytesParser(policy=email.policy.compat32).parsebytes(data)352 353 # We have to wrap parsed.keys() in a set, because in the case of multiple354 # values for a key (a list), the key will appear multiple times in the355 # list of keys, but we're avoiding that by using get_all().356 for name_with_case in frozenset(parsed.keys()):357 # Header names in RFC are case insensitive, so we'll normalize to all358 # lower case to make comparisons easier.359 name = name_with_case.lower()360 361 # We use get_all() here, even for fields that aren't multiple use,362 # because otherwise someone could have e.g. two Name fields, and we363 # would just silently ignore it rather than doing something about it.364 headers = parsed.get_all(name) or []365 366 # The way the email module works when parsing bytes is that it367 # unconditionally decodes the bytes as ascii using the surrogateescape368 # handler. When you pull that data back out (such as with get_all() ),369 # it looks to see if the str has any surrogate escapes, and if it does370 # it wraps it in a Header object instead of returning the string.371 #372 # As such, we'll look for those Header objects, and fix up the encoding.373 value = []374 # Flag if we have run into any issues processing the headers, thus375 # signalling that the data belongs in 'unparsed'.376 valid_encoding = True377 for h in headers:378 # It's unclear if this can return more types than just a Header or379 # a str, so we'll just assert here to make sure.380 assert isinstance(h, (email.header.Header, str))381 382 # If it's a header object, we need to do our little dance to get383 # the real data out of it. In cases where there is invalid data384 # we're going to end up with mojibake, but there's no obvious, good385 # way around that without reimplementing parts of the Header object386 # ourselves.387 #388 # That should be fine since, if mojibacked happens, this key is389 # going into the unparsed dict anyways.390 if isinstance(h, email.header.Header):391 # The Header object stores it's data as chunks, and each chunk392 # can be independently encoded, so we'll need to check each393 # of them.394 chunks: list[tuple[bytes, str | None]] = []395 for binary, _encoding in email.header.decode_header(h):396 try:397 binary.decode("utf8", "strict")398 except UnicodeDecodeError:399 # Enable mojibake.400 encoding = "latin1"401 valid_encoding = False402 else:403 encoding = "utf8"404 chunks.append((binary, encoding))405 406 # Turn our chunks back into a Header object, then let that407 # Header object do the right thing to turn them into a408 # string for us.409 value.append(str(email.header.make_header(chunks)))410 # This is already a string, so just add it.411 else:412 value.append(h)413 414 # We've processed all of our values to get them into a list of str,415 # but we may have mojibake data, in which case this is an unparsed416 # field.417 if not valid_encoding:418 unparsed[name] = value419 continue420 421 raw_name = _EMAIL_TO_RAW_MAPPING.get(name)422 if raw_name is None:423 # This is a bit of a weird situation, we've encountered a key that424 # we don't know what it means, so we don't know whether it's meant425 # to be a list or not.426 #427 # Since we can't really tell one way or another, we'll just leave it428 # as a list, even though it may be a single item list, because that's429 # what makes the most sense for email headers.430 unparsed[name] = value431 continue432 433 # If this is one of our string fields, then we'll check to see if our434 # value is a list of a single item. If it is then we'll assume that435 # it was emitted as a single string, and unwrap the str from inside436 # the list.437 #438 # If it's any other kind of data, then we haven't the faintest clue439 # what we should parse it as, and we have to just add it to our list440 # of unparsed stuff.441 if raw_name in _STRING_FIELDS and len(value) == 1:442 raw[raw_name] = value[0]443 # If this is import_names, we need to special case the empty field444 # case, which converts to an empty list instead of None. We can't let445 # the empty case slip through, as it will fail validation.446 elif raw_name == "import_names" and value == [""]:447 raw[raw_name] = []448 # If this is one of our list of string fields, then we can just assign449 # the value, since email *only* has strings, and our get_all() call450 # above ensures that this is a list.451 elif raw_name in _LIST_FIELDS:452 raw[raw_name] = value453 # Special Case: Keywords454 # The keywords field is implemented in the metadata spec as a str,455 # but it conceptually is a list of strings, and is serialized using456 # ", ".join(keywords), so we'll do some light data massaging to turn457 # this into what it logically is.458 elif raw_name == "keywords" and len(value) == 1:459 raw[raw_name] = _parse_keywords(value[0])460 # Special Case: Project-URL461 # The project urls is implemented in the metadata spec as a list of462 # specially-formatted strings that represent a key and a value, which463 # is fundamentally a mapping, however the email format doesn't support464 # mappings in a sane way, so it was crammed into a list of strings465 # instead.466 #467 # We will do a little light data massaging to turn this into a map as468 # it logically should be.469 elif raw_name == "project_urls":470 try:471 raw[raw_name] = _parse_project_urls(value)472 except KeyError:473 unparsed[name] = value474 # Nothing that we've done has managed to parse this, so it'll just475 # throw it in our unparsable data and move on.476 else:477 unparsed[name] = value478 479 # We need to support getting the Description from the message payload in480 # addition to getting it from the the headers. This does mean, though, there481 # is the possibility of it being set both ways, in which case we put both482 # in 'unparsed' since we don't know which is right.483 try:484 payload = _get_payload(parsed, data)485 except ValueError:486 unparsed.setdefault("description", []).append(487 parsed.get_payload(decode=isinstance(data, bytes)) # type: ignore[call-overload]488 )489 else:490 if payload:491 # Check to see if we've already got a description, if so then both492 # it, and this body move to unparsable.493 if "description" in raw:494 description_header = cast("str", raw.pop("description"))495 unparsed.setdefault("description", []).extend(496 [description_header, payload]497 )498 elif "description" in unparsed:499 unparsed["description"].append(payload)500 else:501 raw["description"] = payload502 503 # We need to cast our `raw` to a metadata, because a TypedDict only support504 # literal key names, but we're computing our key names on purpose, but the505 # way this function is implemented, our `TypedDict` can only have valid key506 # names.507 return cast("RawMetadata", raw), unparsed508 509 510_NOT_FOUND = object()511 512 513# Keep the two values in sync.514_VALID_METADATA_VERSIONS = ["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]515_MetadataVersion = Literal["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]516 517_REQUIRED_ATTRS = frozenset(["metadata_version", "name", "version"])518 519 520class _Validator(Generic[T]):521 """Validate a metadata field.522 523 All _process_*() methods correspond to a core metadata field. The method is524 called with the field's raw value. If the raw value is valid it is returned525 in its "enriched" form (e.g. ``version.Version`` for the ``Version`` field).526 If the raw value is invalid, :exc:`InvalidMetadata` is raised (with a cause527 as appropriate).528 """529 530 name: str531 raw_name: str532 added: _MetadataVersion533 534 def __init__(535 self,536 *,537 added: _MetadataVersion = "1.0",538 ) -> None:539 self.added = added540 541 def __set_name__(self, _owner: Metadata, name: str) -> None:542 self.name = name543 self.raw_name = _RAW_TO_EMAIL_MAPPING[name]544 545 def __get__(self, instance: Metadata, _owner: type[Metadata]) -> T:546 # With Python 3.8, the caching can be replaced with functools.cached_property().547 # No need to check the cache as attribute lookup will resolve into the548 # instance's __dict__ before __get__ is called.549 cache = instance.__dict__550 value = instance._raw.get(self.name)551 552 # To make the _process_* methods easier, we'll check if the value is None553 # and if this field is NOT a required attribute, and if both of those554 # things are true, we'll skip the the converter. This will mean that the555 # converters never have to deal with the None union.556 if self.name in _REQUIRED_ATTRS or value is not None:557 try:558 converter: Callable[[Any], T] = getattr(self, f"_process_{self.name}")559 except AttributeError:560 pass561 else:562 value = converter(value)563 564 cache[self.name] = value565 try:566 del instance._raw[self.name] # type: ignore[misc]567 except KeyError:568 pass569 570 return cast("T", value)571 572 def _invalid_metadata(573 self, msg: str, cause: Exception | None = None574 ) -> InvalidMetadata:575 exc = InvalidMetadata(576 self.raw_name, msg.format_map({"field": repr(self.raw_name)})577 )578 exc.__cause__ = cause579 return exc580 581 def _process_metadata_version(self, value: str) -> _MetadataVersion:582 # Implicitly makes Metadata-Version required.583 if value not in _VALID_METADATA_VERSIONS:584 raise self._invalid_metadata(f"{value!r} is not a valid metadata version")585 return cast("_MetadataVersion", value)586 587 def _process_name(self, value: str) -> str:588 if not value:589 raise self._invalid_metadata("{field} is a required field")590 # Validate the name as a side-effect.591 try:592 utils.canonicalize_name(value, validate=True)593 except utils.InvalidName as exc:594 raise self._invalid_metadata(595 f"{value!r} is invalid for {{field}}", cause=exc596 ) from exc597 else:598 return value599 600 def _process_version(self, value: str) -> version_module.Version:601 if not value:602 raise self._invalid_metadata("{field} is a required field")603 try:604 return version_module.parse(value)605 except version_module.InvalidVersion as exc:606 raise self._invalid_metadata(607 f"{value!r} is invalid for {{field}}", cause=exc608 ) from exc609 610 def _process_summary(self, value: str) -> str:611 """Check the field contains no newlines."""612 if "\n" in value:613 raise self._invalid_metadata("{field} must be a single line")614 return value615 616 def _process_description_content_type(self, value: str) -> str:617 content_types = {"text/plain", "text/x-rst", "text/markdown"}618 message = email.message.EmailMessage()619 message["content-type"] = value620 621 content_type, parameters = (622 # Defaults to `text/plain` if parsing failed.623 message.get_content_type().lower(),624 message["content-type"].params,625 )626 # Check if content-type is valid or defaulted to `text/plain` and thus was627 # not parseable.628 if content_type not in content_types or content_type not in value.lower():629 raise self._invalid_metadata(630 f"{{field}} must be one of {list(content_types)}, not {value!r}"631 )632 633 charset = parameters.get("charset", "UTF-8")634 if charset != "UTF-8":635 raise self._invalid_metadata(636 f"{{field}} can only specify the UTF-8 charset, not {charset!r}"637 )638 639 markdown_variants = {"GFM", "CommonMark"}640 variant = parameters.get("variant", "GFM") # Use an acceptable default.641 if content_type == "text/markdown" and variant not in markdown_variants:642 raise self._invalid_metadata(643 f"valid Markdown variants for {{field}} are {list(markdown_variants)}, "644 f"not {variant!r}",645 )646 return value647 648 def _process_dynamic(self, value: list[str]) -> list[str]:649 for dynamic_field in map(str.lower, value):650 if dynamic_field in {"name", "version", "metadata-version"}:651 raise self._invalid_metadata(652 f"{dynamic_field!r} is not allowed as a dynamic field"653 )654 elif dynamic_field not in _EMAIL_TO_RAW_MAPPING:655 raise self._invalid_metadata(656 f"{dynamic_field!r} is not a valid dynamic field"657 )658 return list(map(str.lower, value))659 660 def _process_provides_extra(661 self,662 value: list[str],663 ) -> list[utils.NormalizedName]:664 normalized_names = []665 try:666 for name in value:667 normalized_names.append(utils.canonicalize_name(name, validate=True))668 except utils.InvalidName as exc:669 raise self._invalid_metadata(670 f"{name!r} is invalid for {{field}}", cause=exc671 ) from exc672 else:673 return normalized_names674 675 def _process_requires_python(self, value: str) -> specifiers.SpecifierSet:676 try:677 return specifiers.SpecifierSet(value)678 except specifiers.InvalidSpecifier as exc:679 raise self._invalid_metadata(680 f"{value!r} is invalid for {{field}}", cause=exc681 ) from exc682 683 def _process_requires_dist(684 self,685 value: list[str],686 ) -> list[requirements.Requirement]:687 reqs = []688 try:689 for req in value:690 reqs.append(requirements.Requirement(req))691 except requirements.InvalidRequirement as exc:692 raise self._invalid_metadata(693 f"{req!r} is invalid for {{field}}", cause=exc694 ) from exc695 else:696 return reqs697 698 def _process_license_expression(self, value: str) -> NormalizedLicenseExpression:699 try:700 return licenses.canonicalize_license_expression(value)701 except ValueError as exc:702 raise self._invalid_metadata(703 f"{value!r} is invalid for {{field}}", cause=exc704 ) from exc705 706 def _process_license_files(self, value: list[str]) -> list[str]:707 paths = []708 for path in value:709 if ".." in path:710 raise self._invalid_metadata(711 f"{path!r} is invalid for {{field}}, "712 "parent directory indicators are not allowed"713 )714 if "*" in path:715 raise self._invalid_metadata(716 f"{path!r} is invalid for {{field}}, paths must be resolved"717 )718 if (719 pathlib.PurePosixPath(path).is_absolute()720 or pathlib.PureWindowsPath(path).is_absolute()721 ):722 raise self._invalid_metadata(723 f"{path!r} is invalid for {{field}}, paths must be relative"724 )725 if pathlib.PureWindowsPath(path).as_posix() != path:726 raise self._invalid_metadata(727 f"{path!r} is invalid for {{field}}, paths must use '/' delimiter"728 )729 paths.append(path)730 return paths731 732 def _process_import_names(self, value: list[str]) -> list[str]:733 for import_name in value:734 name, semicolon, private = import_name.partition(";")735 name = name.rstrip()736 for identifier in name.split("."):737 if not identifier.isidentifier():738 raise self._invalid_metadata(739 f"{name!r} is invalid for {{field}}; "740 f"{identifier!r} is not a valid identifier"741 )742 elif keyword.iskeyword(identifier):743 raise self._invalid_metadata(744 f"{name!r} is invalid for {{field}}; "745 f"{identifier!r} is a keyword"746 )747 if semicolon and private.lstrip() != "private":748 raise self._invalid_metadata(749 f"{import_name!r} is invalid for {{field}}; "750 "the only valid option is 'private'"751 )752 return value753 754 _process_import_namespaces = _process_import_names755 756 757class Metadata:758 """Representation of distribution metadata.759 760 Compared to :class:`RawMetadata`, this class provides objects representing761 metadata fields instead of only using built-in types. Any invalid metadata762 will cause :exc:`InvalidMetadata` to be raised (with a763 :py:attr:`~BaseException.__cause__` attribute as appropriate).764 """765 766 _raw: RawMetadata767 768 @classmethod769 def from_raw(cls, data: RawMetadata, *, validate: bool = True) -> Metadata:770 """Create an instance from :class:`RawMetadata`.771 772 If *validate* is true, all metadata will be validated. All exceptions773 related to validation will be gathered and raised as an :class:`ExceptionGroup`.774 """775 ins = cls()776 ins._raw = data.copy() # Mutations occur due to caching enriched values.777 778 if validate:779 collector = _ErrorCollector()780 metadata_version = None781 with collector.collect(InvalidMetadata):782 metadata_version = ins.metadata_version783 metadata_age = _VALID_METADATA_VERSIONS.index(metadata_version)784 785 # Make sure to check for the fields that are present, the required786 # fields (so their absence can be reported).787 fields_to_check = frozenset(ins._raw) | _REQUIRED_ATTRS788 # Remove fields that have already been checked.789 fields_to_check -= {"metadata_version"}790 791 for key in fields_to_check:792 try:793 if metadata_version:794 # Can't use getattr() as that triggers descriptor protocol which795 # will fail due to no value for the instance argument.796 try:797 field_metadata_version = cls.__dict__[key].added798 except KeyError:799 exc = InvalidMetadata(key, f"unrecognized field: {key!r}")800 collector.error(exc)801 continue802 field_age = _VALID_METADATA_VERSIONS.index(803 field_metadata_version804 )805 if field_age > metadata_age:806 field = _RAW_TO_EMAIL_MAPPING[key]807 exc = InvalidMetadata(808 field,809 f"{field} introduced in metadata version "810 f"{field_metadata_version}, not {metadata_version}",811 )812 collector.error(exc)813 continue814 getattr(ins, key)815 except InvalidMetadata as exc:816 collector.error(exc)817 818 collector.finalize("invalid metadata")819 820 return ins821 822 @classmethod823 def from_email(cls, data: bytes | str, *, validate: bool = True) -> Metadata:824 """Parse metadata from email headers.825 826 If *validate* is true, the metadata will be validated. All exceptions827 related to validation will be gathered and raised as an :class:`ExceptionGroup`.828 """829 raw, unparsed = parse_email(data)830 831 if validate:832 with _ErrorCollector().on_exit("unparsed") as collector:833 for unparsed_key in unparsed:834 if unparsed_key in _EMAIL_TO_RAW_MAPPING:835 message = f"{unparsed_key!r} has invalid data"836 else:837 message = f"unrecognized field: {unparsed_key!r}"838 collector.error(InvalidMetadata(unparsed_key, message))839 840 try:841 return cls.from_raw(raw, validate=validate)842 except ExceptionGroup as exc_group:843 raise ExceptionGroup(844 "invalid or unparsed metadata", exc_group.exceptions845 ) from None846 847 metadata_version: _Validator[_MetadataVersion] = _Validator()848 """:external:ref:`core-metadata-metadata-version`849 (required; validated to be a valid metadata version)"""850 # `name` is not normalized/typed to NormalizedName so as to provide access to851 # the original/raw name.852 name: _Validator[str] = _Validator()853 """:external:ref:`core-metadata-name`854 (required; validated using :func:`~packaging.utils.canonicalize_name` and its855 *validate* parameter)"""856 version: _Validator[version_module.Version] = _Validator()857 """:external:ref:`core-metadata-version` (required)"""858 dynamic: _Validator[list[str] | None] = _Validator(859 added="2.2",860 )861 """:external:ref:`core-metadata-dynamic`862 (validated against core metadata field names and lowercased)"""863 platforms: _Validator[list[str] | None] = _Validator()864 """:external:ref:`core-metadata-platform`"""865 supported_platforms: _Validator[list[str] | None] = _Validator(added="1.1")866 """:external:ref:`core-metadata-supported-platform`"""867 summary: _Validator[str | None] = _Validator()868 """:external:ref:`core-metadata-summary` (validated to contain no newlines)"""869 description: _Validator[str | None] = _Validator() # TODO 2.1: can be in body870 """:external:ref:`core-metadata-description`"""871 description_content_type: _Validator[str | None] = _Validator(added="2.1")872 """:external:ref:`core-metadata-description-content-type` (validated)"""873 keywords: _Validator[list[str] | None] = _Validator()874 """:external:ref:`core-metadata-keywords`"""875 home_page: _Validator[str | None] = _Validator()876 """:external:ref:`core-metadata-home-page`"""877 download_url: _Validator[str | None] = _Validator(added="1.1")878 """:external:ref:`core-metadata-download-url`"""879 author: _Validator[str | None] = _Validator()880 """:external:ref:`core-metadata-author`"""881 author_email: _Validator[str | None] = _Validator()882 """:external:ref:`core-metadata-author-email`"""883 maintainer: _Validator[str | None] = _Validator(added="1.2")884 """:external:ref:`core-metadata-maintainer`"""885 maintainer_email: _Validator[str | None] = _Validator(added="1.2")886 """:external:ref:`core-metadata-maintainer-email`"""887 license: _Validator[str | None] = _Validator()888 """:external:ref:`core-metadata-license`"""889 license_expression: _Validator[NormalizedLicenseExpression | None] = _Validator(890 added="2.4"891 )892 """:external:ref:`core-metadata-license-expression`"""893 license_files: _Validator[list[str] | None] = _Validator(added="2.4")894 """:external:ref:`core-metadata-license-file`"""895 classifiers: _Validator[list[str] | None] = _Validator(added="1.1")896 """:external:ref:`core-metadata-classifier`"""897 requires_dist: _Validator[list[requirements.Requirement] | None] = _Validator(898 added="1.2"899 )900 """:external:ref:`core-metadata-requires-dist`"""901 requires_python: _Validator[specifiers.SpecifierSet | None] = _Validator(902 added="1.2"903 )904 """:external:ref:`core-metadata-requires-python`"""905 # Because `Requires-External` allows for non-PEP 440 version specifiers, we906 # don't do any processing on the values.907 requires_external: _Validator[list[str] | None] = _Validator(added="1.2")908 """:external:ref:`core-metadata-requires-external`"""909 project_urls: _Validator[dict[str, str] | None] = _Validator(added="1.2")910 """:external:ref:`core-metadata-project-url`"""911 # PEP 685 lets us raise an error if an extra doesn't pass `Name` validation912 # regardless of metadata version.913 provides_extra: _Validator[list[utils.NormalizedName] | None] = _Validator(914 added="2.1",915 )916 """:external:ref:`core-metadata-provides-extra`"""917 provides_dist: _Validator[list[str] | None] = _Validator(added="1.2")918 """:external:ref:`core-metadata-provides-dist`"""919 obsoletes_dist: _Validator[list[str] | None] = _Validator(added="1.2")920 """:external:ref:`core-metadata-obsoletes-dist`"""921 import_names: _Validator[list[str] | None] = _Validator(added="2.5")922 """:external:ref:`core-metadata-import-name`"""923 import_namespaces: _Validator[list[str] | None] = _Validator(added="2.5")924 """:external:ref:`core-metadata-import-namespace`"""925 requires: _Validator[list[str] | None] = _Validator(added="1.1")926 """``Requires`` (deprecated)"""927 provides: _Validator[list[str] | None] = _Validator(added="1.1")928 """``Provides`` (deprecated)"""929 obsoletes: _Validator[list[str] | None] = _Validator(added="1.1")930 """``Obsoletes`` (deprecated)"""931 932 def as_rfc822(self) -> RFC822Message:933 """934 Return an RFC822 message with the metadata.935 """936 message = RFC822Message()937 self._write_metadata(message)938 return message939 940 def _write_metadata(self, message: RFC822Message) -> None:941 """942 Return an RFC822 message with the metadata.943 """944 for name, validator in self.__class__.__dict__.items():945 if isinstance(validator, _Validator) and name != "description":946 value = getattr(self, name)947 email_name = _RAW_TO_EMAIL_MAPPING[name]948 if value is not None:949 if email_name == "project-url":950 for label, url in value.items():951 message[email_name] = f"{label}, {url}"952 elif email_name == "keywords":953 message[email_name] = ",".join(value)954 elif email_name == "import-name" and value == []:955 message[email_name] = ""956 elif isinstance(value, list):957 for item in value:958 message[email_name] = str(item)959 else:960 message[email_name] = str(value)961 962 # The description is a special case because it is in the body of the message.963 if self.description is not None:964 message.set_payload(self.description)965 