Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
metadata.py979 linesDownload Raw Back to packaging
1from __future__ import annotations2 3import email.feedparser4import email.header5import email.message6import email.parser7import email.policy8import keyword9import pathlib10import sys11import typing12from typing import (13    Any,14    Callable,15    Generic,16    Literal,17    TypedDict,18    cast,19)20 21from . import licenses, requirements, specifiers, utils22from . import version as version_module23 24if typing.TYPE_CHECKING:25    from .licenses import NormalizedLicenseExpression26 27T = typing.TypeVar("T")28 29 30if sys.version_info >= (3, 11):  # pragma: no cover31    ExceptionGroup = ExceptionGroup  # noqa: F82132else:  # pragma: no cover33 34    class ExceptionGroup(Exception):35        """A minimal implementation of :external:exc:`ExceptionGroup` from Python 3.11.36 37        If :external:exc:`ExceptionGroup` is already defined by Python itself,38        that version is used instead.39        """40 41        message: str42        exceptions: list[Exception]43 44        def __init__(self, message: str, exceptions: list[Exception]) -> None:45            self.message = message46            self.exceptions = exceptions47 48        def __repr__(self) -> str:49            return f"{self.__class__.__name__}({self.message!r}, {self.exceptions!r})"50 51 52class InvalidMetadata(ValueError):53    """A metadata field contains invalid data."""54 55    field: str56    """The name of the field that contains invalid data."""57 58    def __init__(self, field: str, message: str) -> None:59        self.field = field60        super().__init__(message)61 62 63# The RawMetadata class attempts to make as few assumptions about the underlying64# serialization formats as possible. The idea is that as long as a serialization65# formats offer some very basic primitives in *some* way then we can support66# serializing to and from that format.67class RawMetadata(TypedDict, total=False):68    """A dictionary of raw core metadata.69 70    Each field in core metadata maps to a key of this dictionary (when data is71    provided). The key is lower-case and underscores are used instead of dashes72    compared to the equivalent core metadata field. Any core metadata field that73    can be specified multiple times or can hold multiple values in a single74    field have a key with a plural name. See :class:`Metadata` whose attributes75    match the keys of this dictionary.76 77    Core metadata fields that can be specified multiple times are stored as a78    list or dict depending on which is appropriate for the field. Any fields79    which hold multiple values in a single field are stored as a list.80 81    """82 83    # Metadata 1.0 - PEP 24184    metadata_version: str85    name: str86    version: str87    platforms: list[str]88    summary: str89    description: str90    keywords: list[str]91    home_page: str92    author: str93    author_email: str94    license: str95 96    # Metadata 1.1 - PEP 31497    supported_platforms: list[str]98    download_url: str99    classifiers: list[str]100    requires: list[str]101    provides: list[str]102    obsoletes: list[str]103 104    # Metadata 1.2 - PEP 345105    maintainer: str106    maintainer_email: str107    requires_dist: list[str]108    provides_dist: list[str]109    obsoletes_dist: list[str]110    requires_python: str111    requires_external: list[str]112    project_urls: dict[str, str]113 114    # Metadata 2.0115    # PEP 426 attempted to completely revamp the metadata format116    # but got stuck without ever being able to build consensus on117    # it and ultimately ended up withdrawn.118    #119    # However, a number of tools had started emitting METADATA with120    # `2.0` Metadata-Version, so for historical reasons, this version121    # was skipped.122 123    # Metadata 2.1 - PEP 566124    description_content_type: str125    provides_extra: list[str]126 127    # Metadata 2.2 - PEP 643128    dynamic: list[str]129 130    # Metadata 2.3 - PEP 685131    # No new fields were added in PEP 685, just some edge case were132    # tightened up to provide better interoperability.133 134    # Metadata 2.4 - PEP 639135    license_expression: str136    license_files: list[str]137 138    # Metadata 2.5 - PEP 794139    import_names: list[str]140    import_namespaces: list[str]141 142 143# 'keywords' is special as it's a string in the core metadata spec, but we144# represent it as a list.145_STRING_FIELDS = {146    "author",147    "author_email",148    "description",149    "description_content_type",150    "download_url",151    "home_page",152    "license",153    "license_expression",154    "maintainer",155    "maintainer_email",156    "metadata_version",157    "name",158    "requires_python",159    "summary",160    "version",161}162 163_LIST_FIELDS = {164    "classifiers",165    "dynamic",166    "license_files",167    "obsoletes",168    "obsoletes_dist",169    "platforms",170    "provides",171    "provides_dist",172    "provides_extra",173    "requires",174    "requires_dist",175    "requires_external",176    "supported_platforms",177    "import_names",178    "import_namespaces",179}180 181_DICT_FIELDS = {182    "project_urls",183}184 185 186def _parse_keywords(data: str) -> list[str]:187    """Split a string of comma-separated keywords into a list of keywords."""188    return [k.strip() for k in data.split(",")]189 190 191def _parse_project_urls(data: list[str]) -> dict[str, str]:192    """Parse a list of label/URL string pairings separated by a comma."""193    urls = {}194    for pair in data:195        # Our logic is slightly tricky here as we want to try and do196        # *something* reasonable with malformed data.197        #198        # The main thing that we have to worry about, is data that does199        # not have a ',' at all to split the label from the Value. There200        # isn't a singular right answer here, and we will fail validation201        # later on (if the caller is validating) so it doesn't *really*202        # matter, but since the missing value has to be an empty str203        # and our return value is dict[str, str], if we let the key204        # be the missing value, then they'd have multiple '' values that205        # overwrite each other in a accumulating dict.206        #207        # The other potential issue is that it's possible to have the208        # same label multiple times in the metadata, with no solid "right"209        # answer with what to do in that case. As such, we'll do the only210        # thing we can, which is treat the field as unparsable and add it211        # to our list of unparsed fields.212        #213        # TODO: The spec doesn't say anything about if the keys should be214        #       considered case sensitive or not... logically they should215        #       be case-preserving and case-insensitive, but doing that216        #       would open up more cases where we might have duplicate217        #       entries.218        label, _, url = (s.strip() for s in pair.partition(","))219 220        if label in urls:221            # The label already exists in our set of urls, so this field222            # is unparsable, and we can just add the whole thing to our223            # unparsable data and stop processing it.224            raise KeyError("duplicate labels in project urls")225        urls[label] = url226 227    return urls228 229 230def _get_payload(msg: email.message.Message, source: bytes | str) -> str:231    """Get the body of the message."""232    # If our source is a str, then our caller has managed encodings for us,233    # and we don't need to deal with it.234    if isinstance(source, str):235        payload = msg.get_payload()236        assert isinstance(payload, str)237        return payload238    # If our source is a bytes, then we're managing the encoding and we need239    # to deal with it.240    else:241        bpayload = msg.get_payload(decode=True)242        assert isinstance(bpayload, bytes)243        try:244            return bpayload.decode("utf8", "strict")245        except UnicodeDecodeError as exc:246            raise ValueError("payload in an invalid encoding") from exc247 248 249# The various parse_FORMAT functions here are intended to be as lenient as250# possible in their parsing, while still returning a correctly typed251# RawMetadata.252#253# To aid in this, we also generally want to do as little touching of the254# data as possible, except where there are possibly some historic holdovers255# that make valid data awkward to work with.256#257# While this is a lower level, intermediate format than our ``Metadata``258# class, some light touch ups can make a massive difference in usability.259 260# Map METADATA fields to RawMetadata.261_EMAIL_TO_RAW_MAPPING = {262    "author": "author",263    "author-email": "author_email",264    "classifier": "classifiers",265    "description": "description",266    "description-content-type": "description_content_type",267    "download-url": "download_url",268    "dynamic": "dynamic",269    "home-page": "home_page",270    "import-name": "import_names",271    "import-namespace": "import_namespaces",272    "keywords": "keywords",273    "license": "license",274    "license-expression": "license_expression",275    "license-file": "license_files",276    "maintainer": "maintainer",277    "maintainer-email": "maintainer_email",278    "metadata-version": "metadata_version",279    "name": "name",280    "obsoletes": "obsoletes",281    "obsoletes-dist": "obsoletes_dist",282    "platform": "platforms",283    "project-url": "project_urls",284    "provides": "provides",285    "provides-dist": "provides_dist",286    "provides-extra": "provides_extra",287    "requires": "requires",288    "requires-dist": "requires_dist",289    "requires-external": "requires_external",290    "requires-python": "requires_python",291    "summary": "summary",292    "supported-platform": "supported_platforms",293    "version": "version",294}295_RAW_TO_EMAIL_MAPPING = {raw: email for email, raw in _EMAIL_TO_RAW_MAPPING.items()}296 297 298# This class is for writing RFC822 messages299class RFC822Policy(email.policy.EmailPolicy):300    """301    This is :class:`email.policy.EmailPolicy`, but with a simple ``header_store_parse``302    implementation that handles multi-line values, and some nice defaults.303    """304 305    utf8 = True306    mangle_from_ = False307    max_line_length = 0308 309    def header_store_parse(self, name: str, value: str) -> tuple[str, str]:310        size = len(name) + 2311        value = value.replace("\n", "\n" + " " * size)312        return (name, value)313 314 315# This class is for writing RFC822 messages316class RFC822Message(email.message.EmailMessage):317    """318    This is :class:`email.message.EmailMessage` with two small changes: it defaults to319    our `RFC822Policy`, and it correctly writes unicode when being called320    with `bytes()`.321    """322 323    def __init__(self) -> None:324        super().__init__(policy=RFC822Policy())325 326    def as_bytes(327        self, unixfrom: bool = False, policy: email.policy.Policy | None = None328    ) -> bytes:329        """330        Return the bytes representation of the message.331 332        This handles unicode encoding.333        """334        return self.as_string(unixfrom, policy=policy).encode("utf-8")335 336 337def parse_email(data: bytes | str) -> tuple[RawMetadata, dict[str, list[str]]]:338    """Parse a distribution's metadata stored as email headers (e.g. from ``METADATA``).339 340    This function returns a two-item tuple of dicts. The first dict is of341    recognized fields from the core metadata specification. Fields that can be342    parsed and translated into Python's built-in types are converted343    appropriately. All other fields are left as-is. Fields that are allowed to344    appear multiple times are stored as lists.345 346    The second dict contains all other fields from the metadata. This includes347    any unrecognized fields. It also includes any fields which are expected to348    be parsed into a built-in type but were not formatted appropriately. Finally,349    any fields that are expected to appear only once but are repeated are350    included in this dict.351 352    """353    raw: dict[str, str | list[str] | dict[str, str]] = {}354    unparsed: dict[str, list[str]] = {}355 356    if isinstance(data, str):357        parsed = email.parser.Parser(policy=email.policy.compat32).parsestr(data)358    else:359        parsed = email.parser.BytesParser(policy=email.policy.compat32).parsebytes(data)360 361    # We have to wrap parsed.keys() in a set, because in the case of multiple362    # values for a key (a list), the key will appear multiple times in the363    # list of keys, but we're avoiding that by using get_all().364    for name_with_case in frozenset(parsed.keys()):365        # Header names in RFC are case insensitive, so we'll normalize to all366        # lower case to make comparisons easier.367        name = name_with_case.lower()368 369        # We use get_all() here, even for fields that aren't multiple use,370        # because otherwise someone could have e.g. two Name fields, and we371        # would just silently ignore it rather than doing something about it.372        headers = parsed.get_all(name) or []373 374        # The way the email module works when parsing bytes is that it375        # unconditionally decodes the bytes as ascii using the surrogateescape376        # handler. When you pull that data back out (such as with get_all() ),377        # it looks to see if the str has any surrogate escapes, and if it does378        # it wraps it in a Header object instead of returning the string.379        #380        # As such, we'll look for those Header objects, and fix up the encoding.381        value = []382        # Flag if we have run into any issues processing the headers, thus383        # signalling that the data belongs in 'unparsed'.384        valid_encoding = True385        for h in headers:386            # It's unclear if this can return more types than just a Header or387            # a str, so we'll just assert here to make sure.388            assert isinstance(h, (email.header.Header, str))389 390            # If it's a header object, we need to do our little dance to get391            # the real data out of it. In cases where there is invalid data392            # we're going to end up with mojibake, but there's no obvious, good393            # way around that without reimplementing parts of the Header object394            # ourselves.395            #396            # That should be fine since, if mojibacked happens, this key is397            # going into the unparsed dict anyways.398            if isinstance(h, email.header.Header):399                # The Header object stores it's data as chunks, and each chunk400                # can be independently encoded, so we'll need to check each401                # of them.402                chunks: list[tuple[bytes, str | None]] = []403                for binary, _encoding in email.header.decode_header(h):404                    try:405                        binary.decode("utf8", "strict")406                    except UnicodeDecodeError:407                        # Enable mojibake.408                        encoding = "latin1"409                        valid_encoding = False410                    else:411                        encoding = "utf8"412                    chunks.append((binary, encoding))413 414                # Turn our chunks back into a Header object, then let that415                # Header object do the right thing to turn them into a416                # string for us.417                value.append(str(email.header.make_header(chunks)))418            # This is already a string, so just add it.419            else:420                value.append(h)421 422        # We've processed all of our values to get them into a list of str,423        # but we may have mojibake data, in which case this is an unparsed424        # field.425        if not valid_encoding:426            unparsed[name] = value427            continue428 429        raw_name = _EMAIL_TO_RAW_MAPPING.get(name)430        if raw_name is None:431            # This is a bit of a weird situation, we've encountered a key that432            # we don't know what it means, so we don't know whether it's meant433            # to be a list or not.434            #435            # Since we can't really tell one way or another, we'll just leave it436            # as a list, even though it may be a single item list, because that's437            # what makes the most sense for email headers.438            unparsed[name] = value439            continue440 441        # If this is one of our string fields, then we'll check to see if our442        # value is a list of a single item. If it is then we'll assume that443        # it was emitted as a single string, and unwrap the str from inside444        # the list.445        #446        # If it's any other kind of data, then we haven't the faintest clue447        # what we should parse it as, and we have to just add it to our list448        # of unparsed stuff.449        if raw_name in _STRING_FIELDS and len(value) == 1:450            raw[raw_name] = value[0]451        # If this is import_names, we need to special case the empty field452        # case, which converts to an empty list instead of None. We can't let453        # the empty case slip through, as it will fail validation.454        elif raw_name == "import_names" and value == [""]:455            raw[raw_name] = []456        # If this is one of our list of string fields, then we can just assign457        # the value, since email *only* has strings, and our get_all() call458        # above ensures that this is a list.459        elif raw_name in _LIST_FIELDS:460            raw[raw_name] = value461        # Special Case: Keywords462        # The keywords field is implemented in the metadata spec as a str,463        # but it conceptually is a list of strings, and is serialized using464        # ", ".join(keywords), so we'll do some light data massaging to turn465        # this into what it logically is.466        elif raw_name == "keywords" and len(value) == 1:467            raw[raw_name] = _parse_keywords(value[0])468        # Special Case: Project-URL469        # The project urls is implemented in the metadata spec as a list of470        # specially-formatted strings that represent a key and a value, which471        # is fundamentally a mapping, however the email format doesn't support472        # mappings in a sane way, so it was crammed into a list of strings473        # instead.474        #475        # We will do a little light data massaging to turn this into a map as476        # it logically should be.477        elif raw_name == "project_urls":478            try:479                raw[raw_name] = _parse_project_urls(value)480            except KeyError:481                unparsed[name] = value482        # Nothing that we've done has managed to parse this, so it'll just483        # throw it in our unparsable data and move on.484        else:485            unparsed[name] = value486 487    # We need to support getting the Description from the message payload in488    # addition to getting it from the the headers. This does mean, though, there489    # is the possibility of it being set both ways, in which case we put both490    # in 'unparsed' since we don't know which is right.491    try:492        payload = _get_payload(parsed, data)493    except ValueError:494        unparsed.setdefault("description", []).append(495            parsed.get_payload(decode=isinstance(data, bytes))  # type: ignore[call-overload]496        )497    else:498        if payload:499            # Check to see if we've already got a description, if so then both500            # it, and this body move to unparsable.501            if "description" in raw:502                description_header = cast("str", raw.pop("description"))503                unparsed.setdefault("description", []).extend(504                    [description_header, payload]505                )506            elif "description" in unparsed:507                unparsed["description"].append(payload)508            else:509                raw["description"] = payload510 511    # We need to cast our `raw` to a metadata, because a TypedDict only support512    # literal key names, but we're computing our key names on purpose, but the513    # way this function is implemented, our `TypedDict` can only have valid key514    # names.515    return cast("RawMetadata", raw), unparsed516 517 518_NOT_FOUND = object()519 520 521# Keep the two values in sync.522_VALID_METADATA_VERSIONS = ["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]523_MetadataVersion = Literal["1.0", "1.1", "1.2", "2.1", "2.2", "2.3", "2.4", "2.5"]524 525_REQUIRED_ATTRS = frozenset(["metadata_version", "name", "version"])526 527 528class _Validator(Generic[T]):529    """Validate a metadata field.530 531    All _process_*() methods correspond to a core metadata field. The method is532    called with the field's raw value. If the raw value is valid it is returned533    in its "enriched" form (e.g. ``version.Version`` for the ``Version`` field).534    If the raw value is invalid, :exc:`InvalidMetadata` is raised (with a cause535    as appropriate).536    """537 538    name: str539    raw_name: str540    added: _MetadataVersion541 542    def __init__(543        self,544        *,545        added: _MetadataVersion = "1.0",546    ) -> None:547        self.added = added548 549    def __set_name__(self, _owner: Metadata, name: str) -> None:550        self.name = name551        self.raw_name = _RAW_TO_EMAIL_MAPPING[name]552 553    def __get__(self, instance: Metadata, _owner: type[Metadata]) -> T:554        # With Python 3.8, the caching can be replaced with functools.cached_property().555        # No need to check the cache as attribute lookup will resolve into the556        # instance's __dict__ before __get__ is called.557        cache = instance.__dict__558        value = instance._raw.get(self.name)559 560        # To make the _process_* methods easier, we'll check if the value is None561        # and if this field is NOT a required attribute, and if both of those562        # things are true, we'll skip the the converter. This will mean that the563        # converters never have to deal with the None union.564        if self.name in _REQUIRED_ATTRS or value is not None:565            try:566                converter: Callable[[Any], T] = getattr(self, f"_process_{self.name}")567            except AttributeError:568                pass569            else:570                value = converter(value)571 572        cache[self.name] = value573        try:574            del instance._raw[self.name]  # type: ignore[misc]575        except KeyError:576            pass577 578        return cast("T", value)579 580    def _invalid_metadata(581        self, msg: str, cause: Exception | None = None582    ) -> InvalidMetadata:583        exc = InvalidMetadata(584            self.raw_name, msg.format_map({"field": repr(self.raw_name)})585        )586        exc.__cause__ = cause587        return exc588 589    def _process_metadata_version(self, value: str) -> _MetadataVersion:590        # Implicitly makes Metadata-Version required.591        if value not in _VALID_METADATA_VERSIONS:592            raise self._invalid_metadata(f"{value!r} is not a valid metadata version")593        return cast("_MetadataVersion", value)594 595    def _process_name(self, value: str) -> str:596        if not value:597            raise self._invalid_metadata("{field} is a required field")598        # Validate the name as a side-effect.599        try:600            utils.canonicalize_name(value, validate=True)601        except utils.InvalidName as exc:602            raise self._invalid_metadata(603                f"{value!r} is invalid for {{field}}", cause=exc604            ) from exc605        else:606            return value607 608    def _process_version(self, value: str) -> version_module.Version:609        if not value:610            raise self._invalid_metadata("{field} is a required field")611        try:612            return version_module.parse(value)613        except version_module.InvalidVersion as exc:614            raise self._invalid_metadata(615                f"{value!r} is invalid for {{field}}", cause=exc616            ) from exc617 618    def _process_summary(self, value: str) -> str:619        """Check the field contains no newlines."""620        if "\n" in value:621            raise self._invalid_metadata("{field} must be a single line")622        return value623 624    def _process_description_content_type(self, value: str) -> str:625        content_types = {"text/plain", "text/x-rst", "text/markdown"}626        message = email.message.EmailMessage()627        message["content-type"] = value628 629        content_type, parameters = (630            # Defaults to `text/plain` if parsing failed.631            message.get_content_type().lower(),632            message["content-type"].params,633        )634        # Check if content-type is valid or defaulted to `text/plain` and thus was635        # not parseable.636        if content_type not in content_types or content_type not in value.lower():637            raise self._invalid_metadata(638                f"{{field}} must be one of {list(content_types)}, not {value!r}"639            )640 641        charset = parameters.get("charset", "UTF-8")642        if charset != "UTF-8":643            raise self._invalid_metadata(644                f"{{field}} can only specify the UTF-8 charset, not {list(charset)}"645            )646 647        markdown_variants = {"GFM", "CommonMark"}648        variant = parameters.get("variant", "GFM")  # Use an acceptable default.649        if content_type == "text/markdown" and variant not in markdown_variants:650            raise self._invalid_metadata(651                f"valid Markdown variants for {{field}} are {list(markdown_variants)}, "652                f"not {variant!r}",653            )654        return value655 656    def _process_dynamic(self, value: list[str]) -> list[str]:657        for dynamic_field in map(str.lower, value):658            if dynamic_field in {"name", "version", "metadata-version"}:659                raise self._invalid_metadata(660                    f"{dynamic_field!r} is not allowed as a dynamic field"661                )662            elif dynamic_field not in _EMAIL_TO_RAW_MAPPING:663                raise self._invalid_metadata(664                    f"{dynamic_field!r} is not a valid dynamic field"665                )666        return list(map(str.lower, value))667 668    def _process_provides_extra(669        self,670        value: list[str],671    ) -> list[utils.NormalizedName]:672        normalized_names = []673        try:674            for name in value:675                normalized_names.append(utils.canonicalize_name(name, validate=True))676        except utils.InvalidName as exc:677            raise self._invalid_metadata(678                f"{name!r} is invalid for {{field}}", cause=exc679            ) from exc680        else:681            return normalized_names682 683    def _process_requires_python(self, value: str) -> specifiers.SpecifierSet:684        try:685            return specifiers.SpecifierSet(value)686        except specifiers.InvalidSpecifier as exc:687            raise self._invalid_metadata(688                f"{value!r} is invalid for {{field}}", cause=exc689            ) from exc690 691    def _process_requires_dist(692        self,693        value: list[str],694    ) -> list[requirements.Requirement]:695        reqs = []696        try:697            for req in value:698                reqs.append(requirements.Requirement(req))699        except requirements.InvalidRequirement as exc:700            raise self._invalid_metadata(701                f"{req!r} is invalid for {{field}}", cause=exc702            ) from exc703        else:704            return reqs705 706    def _process_license_expression(self, value: str) -> NormalizedLicenseExpression:707        try:708            return licenses.canonicalize_license_expression(value)709        except ValueError as exc:710            raise self._invalid_metadata(711                f"{value!r} is invalid for {{field}}", cause=exc712            ) from exc713 714    def _process_license_files(self, value: list[str]) -> list[str]:715        paths = []716        for path in value:717            if ".." in path:718                raise self._invalid_metadata(719                    f"{path!r} is invalid for {{field}}, "720                    "parent directory indicators are not allowed"721                )722            if "*" in path:723                raise self._invalid_metadata(724                    f"{path!r} is invalid for {{field}}, paths must be resolved"725                )726            if (727                pathlib.PurePosixPath(path).is_absolute()728                or pathlib.PureWindowsPath(path).is_absolute()729            ):730                raise self._invalid_metadata(731                    f"{path!r} is invalid for {{field}}, paths must be relative"732                )733            if pathlib.PureWindowsPath(path).as_posix() != path:734                raise self._invalid_metadata(735                    f"{path!r} is invalid for {{field}}, paths must use '/' delimiter"736                )737            paths.append(path)738        return paths739 740    def _process_import_names(self, value: list[str]) -> list[str]:741        for import_name in value:742            name, semicolon, private = import_name.partition(";")743            name = name.rstrip()744            for identifier in name.split("."):745                if not identifier.isidentifier():746                    raise self._invalid_metadata(747                        f"{name!r} is invalid for {{field}}; "748                        f"{identifier!r} is not a valid identifier"749                    )750                elif keyword.iskeyword(identifier):751                    raise self._invalid_metadata(752                        f"{name!r} is invalid for {{field}}; "753                        f"{identifier!r} is a keyword"754                    )755            if semicolon and private.lstrip() != "private":756                raise self._invalid_metadata(757                    f"{import_name!r} is invalid for {{field}}; "758                    "the only valid option is 'private'"759                )760        return value761 762    _process_import_namespaces = _process_import_names763 764 765class Metadata:766    """Representation of distribution metadata.767 768    Compared to :class:`RawMetadata`, this class provides objects representing769    metadata fields instead of only using built-in types. Any invalid metadata770    will cause :exc:`InvalidMetadata` to be raised (with a771    :py:attr:`~BaseException.__cause__` attribute as appropriate).772    """773 774    _raw: RawMetadata775 776    @classmethod777    def from_raw(cls, data: RawMetadata, *, validate: bool = True) -> Metadata:778        """Create an instance from :class:`RawMetadata`.779 780        If *validate* is true, all metadata will be validated. All exceptions781        related to validation will be gathered and raised as an :class:`ExceptionGroup`.782        """783        ins = cls()784        ins._raw = data.copy()  # Mutations occur due to caching enriched values.785 786        if validate:787            exceptions: list[Exception] = []788            try:789                metadata_version = ins.metadata_version790                metadata_age = _VALID_METADATA_VERSIONS.index(metadata_version)791            except InvalidMetadata as metadata_version_exc:792                exceptions.append(metadata_version_exc)793                metadata_version = None794 795            # Make sure to check for the fields that are present, the required796            # fields (so their absence can be reported).797            fields_to_check = frozenset(ins._raw) | _REQUIRED_ATTRS798            # Remove fields that have already been checked.799            fields_to_check -= {"metadata_version"}800 801            for key in fields_to_check:802                try:803                    if metadata_version:804                        # Can't use getattr() as that triggers descriptor protocol which805                        # will fail due to no value for the instance argument.806                        try:807                            field_metadata_version = cls.__dict__[key].added808                        except KeyError:809                            exc = InvalidMetadata(key, f"unrecognized field: {key!r}")810                            exceptions.append(exc)811                            continue812                        field_age = _VALID_METADATA_VERSIONS.index(813                            field_metadata_version814                        )815                        if field_age > metadata_age:816                            field = _RAW_TO_EMAIL_MAPPING[key]817                            exc = InvalidMetadata(818                                field,819                                f"{field} introduced in metadata version "820                                f"{field_metadata_version}, not {metadata_version}",821                            )822                            exceptions.append(exc)823                            continue824                    getattr(ins, key)825                except InvalidMetadata as exc:826                    exceptions.append(exc)827 828            if exceptions:829                raise ExceptionGroup("invalid metadata", exceptions)830 831        return ins832 833    @classmethod834    def from_email(cls, data: bytes | str, *, validate: bool = True) -> Metadata:835        """Parse metadata from email headers.836 837        If *validate* is true, the metadata will be validated. All exceptions838        related to validation will be gathered and raised as an :class:`ExceptionGroup`.839        """840        raw, unparsed = parse_email(data)841 842        if validate:843            exceptions: list[Exception] = []844            for unparsed_key in unparsed:845                if unparsed_key in _EMAIL_TO_RAW_MAPPING:846                    message = f"{unparsed_key!r} has invalid data"847                else:848                    message = f"unrecognized field: {unparsed_key!r}"849                exceptions.append(InvalidMetadata(unparsed_key, message))850 851            if exceptions:852                raise ExceptionGroup("unparsed", exceptions)853 854        try:855            return cls.from_raw(raw, validate=validate)856        except ExceptionGroup as exc_group:857            raise ExceptionGroup(858                "invalid or unparsed metadata", exc_group.exceptions859            ) from None860 861    metadata_version: _Validator[_MetadataVersion] = _Validator()862    """:external:ref:`core-metadata-metadata-version`863    (required; validated to be a valid metadata version)"""864    # `name` is not normalized/typed to NormalizedName so as to provide access to865    # the original/raw name.866    name: _Validator[str] = _Validator()867    """:external:ref:`core-metadata-name`868    (required; validated using :func:`~packaging.utils.canonicalize_name` and its869    *validate* parameter)"""870    version: _Validator[version_module.Version] = _Validator()871    """:external:ref:`core-metadata-version` (required)"""872    dynamic: _Validator[list[str] | None] = _Validator(873        added="2.2",874    )875    """:external:ref:`core-metadata-dynamic`876    (validated against core metadata field names and lowercased)"""877    platforms: _Validator[list[str] | None] = _Validator()878    """:external:ref:`core-metadata-platform`"""879    supported_platforms: _Validator[list[str] | None] = _Validator(added="1.1")880    """:external:ref:`core-metadata-supported-platform`"""881    summary: _Validator[str | None] = _Validator()882    """:external:ref:`core-metadata-summary` (validated to contain no newlines)"""883    description: _Validator[str | None] = _Validator()  # TODO 2.1: can be in body884    """:external:ref:`core-metadata-description`"""885    description_content_type: _Validator[str | None] = _Validator(added="2.1")886    """:external:ref:`core-metadata-description-content-type` (validated)"""887    keywords: _Validator[list[str] | None] = _Validator()888    """:external:ref:`core-metadata-keywords`"""889    home_page: _Validator[str | None] = _Validator()890    """:external:ref:`core-metadata-home-page`"""891    download_url: _Validator[str | None] = _Validator(added="1.1")892    """:external:ref:`core-metadata-download-url`"""893    author: _Validator[str | None] = _Validator()894    """:external:ref:`core-metadata-author`"""895    author_email: _Validator[str | None] = _Validator()896    """:external:ref:`core-metadata-author-email`"""897    maintainer: _Validator[str | None] = _Validator(added="1.2")898    """:external:ref:`core-metadata-maintainer`"""899    maintainer_email: _Validator[str | None] = _Validator(added="1.2")900    """:external:ref:`core-metadata-maintainer-email`"""901    license: _Validator[str | None] = _Validator()902    """:external:ref:`core-metadata-license`"""903    license_expression: _Validator[NormalizedLicenseExpression | None] = _Validator(904        added="2.4"905    )906    """:external:ref:`core-metadata-license-expression`"""907    license_files: _Validator[list[str] | None] = _Validator(added="2.4")908    """:external:ref:`core-metadata-license-file`"""909    classifiers: _Validator[list[str] | None] = _Validator(added="1.1")910    """:external:ref:`core-metadata-classifier`"""911    requires_dist: _Validator[list[requirements.Requirement] | None] = _Validator(912        added="1.2"913    )914    """:external:ref:`core-metadata-requires-dist`"""915    requires_python: _Validator[specifiers.SpecifierSet | None] = _Validator(916        added="1.2"917    )918    """:external:ref:`core-metadata-requires-python`"""919    # Because `Requires-External` allows for non-PEP 440 version specifiers, we920    # don't do any processing on the values.921    requires_external: _Validator[list[str] | None] = _Validator(added="1.2")922    """:external:ref:`core-metadata-requires-external`"""923    project_urls: _Validator[dict[str, str] | None] = _Validator(added="1.2")924    """:external:ref:`core-metadata-project-url`"""925    # PEP 685 lets us raise an error if an extra doesn't pass `Name` validation926    # regardless of metadata version.927    provides_extra: _Validator[list[utils.NormalizedName] | None] = _Validator(928        added="2.1",929    )930    """:external:ref:`core-metadata-provides-extra`"""931    provides_dist: _Validator[list[str] | None] = _Validator(added="1.2")932    """:external:ref:`core-metadata-provides-dist`"""933    obsoletes_dist: _Validator[list[str] | None] = _Validator(added="1.2")934    """:external:ref:`core-metadata-obsoletes-dist`"""935    import_names: _Validator[list[str] | None] = _Validator(added="2.5")936    """:external:ref:`core-metadata-import-name`"""937    import_namespaces: _Validator[list[str] | None] = _Validator(added="2.5")938    """:external:ref:`core-metadata-import-namespace`"""939    requires: _Validator[list[str] | None] = _Validator(added="1.1")940    """``Requires`` (deprecated)"""941    provides: _Validator[list[str] | None] = _Validator(added="1.1")942    """``Provides`` (deprecated)"""943    obsoletes: _Validator[list[str] | None] = _Validator(added="1.1")944    """``Obsoletes`` (deprecated)"""945 946    def as_rfc822(self) -> RFC822Message:947        """948        Return an RFC822 message with the metadata.949        """950        message = RFC822Message()951        self._write_metadata(message)952        return message953 954    def _write_metadata(self, message: RFC822Message) -> None:955        """956        Return an RFC822 message with the metadata.957        """958        for name, validator in self.__class__.__dict__.items():959            if isinstance(validator, _Validator) and name != "description":960                value = getattr(self, name)961                email_name = _RAW_TO_EMAIL_MAPPING[name]962                if value is not None:963                    if email_name == "project-url":964                        for label, url in value.items():965                            message[email_name] = f"{label}, {url}"966                    elif email_name == "keywords":967                        message[email_name] = ",".join(value)968                    elif email_name == "import-name" and value == []:969                        message[email_name] = ""970                    elif isinstance(value, list):971                        for item in value:972                            message[email_name] = str(item)973                    else:974                        message[email_name] = str(value)975 976        # The description is a special case because it is in the body of the message.977        if self.description is not None:978            message.set_payload(self.description)979 
codekingpro/portable-devtools · Team Ai