forked from pulp/pulp_python
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathutils.py
More file actions
314 lines (265 loc) · 11.4 KB
/
Copy pathutils.py
File metadata and controls
314 lines (265 loc) · 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
import pkginfo
import shutil
import tempfile
import json
from collections import defaultdict
from django.core.files.storage import default_storage as storage
from django.conf import settings
from jinja2 import Template
from packaging.utils import canonicalize_name
from packaging.version import parse
PYPI_LAST_SERIAL = "X-PYPI-LAST-SERIAL"
"""TODO This serial constant is temporary until Python repositories implements serials"""
PYPI_SERIAL_CONSTANT = 1000000000
simple_index_template = """<!DOCTYPE html>
<html>
<head>
<title>Simple Index</title>
<meta name="api-version" value="2" />
</head>
<body>
{% for name, canonical_name in projects %}
<a href="{{ canonical_name }}/">{{ name }}</a><br/>
{% endfor %}
</body>
</html>
"""
simple_detail_template = """<!DOCTYPE html>
<html>
<head>
<title>Links for {{ project_name }}</title>
<meta name="api-version" value="2" />
</head>
<body>
<h1>Links for {{ project_name }}</h1>
{% for name, path, sha256 in project_packages %}
<a href="{{ path }}#sha256={{ sha256 }}" rel="internal">{{ name }}</a><br/>
{% endfor %}
</body>
</html>
"""
DIST_EXTENSIONS = {
".whl": "bdist_wheel",
".exe": "bdist_wininst",
".egg": "bdist_egg",
".tar.bz2": "sdist",
".tar.gz": "sdist",
".zip": "sdist",
}
DIST_TYPES = {
"bdist_wheel": pkginfo.Wheel,
"bdist_wininst": pkginfo.Distribution,
"bdist_egg": pkginfo.BDist,
"sdist": pkginfo.SDist,
}
def parse_project_metadata(project):
"""
Create a dictionary of python project metadata.
Args:
project (dict): Metadata relevant to the entire Python project
Returns:
dictionary: of python project metadata
"""
package = {}
package['name'] = project.get('name') or ""
package['metadata_version'] = project.get('metadata_version') or ""
package['summary'] = project.get('summary') or ""
package['description'] = project.get('description') or ""
package['keywords'] = project.get('keywords') or ""
package['home_page'] = project.get('home_page') or ""
package['download_url'] = project.get('download_url') or ""
package['author'] = project.get('author') or ""
package['author_email'] = project.get('author_email') or ""
package['maintainer'] = project.get('maintainer') or ""
package['maintainer_email'] = project.get('maintainer_email') or ""
package['license'] = project.get('license') or ""
package['project_url'] = project.get('project_url') or ""
package['platform'] = project.get('platform') or ""
package['supported_platform'] = project.get('supported_platform') or ""
package['requires_dist'] = json.dumps(project.get('requires_dist', []))
package['provides_dist'] = json.dumps(project.get('provides_dist', []))
package['obsoletes_dist'] = json.dumps(project.get('obsoletes_dist', []))
package['requires_external'] = json.dumps(project.get('requires_external', []))
package['classifiers'] = json.dumps(project.get('classifiers', []))
package['project_urls'] = json.dumps(project.get('project_urls', {}))
package['description_content_type'] = project.get('description_content_type') or ""
return package
def parse_metadata(project, version, distribution):
"""
Extract metadata from a distribution.
Create a dictionary of metadata needed to create a PythonContentUnit from
the project, version, and distribution metadata.
Args:
project (dict): Metadata relevant to the entire Python project
version (string): Version of distribution
distribution (dict): Metadata of a single Python distribution
Returns:
dictionary: of useful python metadata
"""
package = {}
package['filename'] = distribution.get('filename') or ""
package['packagetype'] = distribution.get('packagetype') or ""
package['version'] = version
package['url'] = distribution.get('url') or ""
package['sha256'] = distribution.get('digests', {}).get('sha256') or ""
package['python_version'] = distribution.get('python_version') or ""
package['requires_python'] = distribution.get('requires_python') or ""
package.update(parse_project_metadata(project))
return package
def get_project_metadata_from_artifact(filename, artifact):
"""
Gets the metadata of a Python Package.
Raises ValueError if filename has an unsupported extension
"""
extensions = list(DIST_EXTENSIONS.keys())
# Iterate through extensions since splitext does not support things like .tar.gz
# If no supported extension is found, ValueError is raised here
pkg_type_index = [filename.endswith(ext) for ext in extensions].index(True)
packagetype = DIST_EXTENSIONS[extensions[pkg_type_index]]
# Copy file to a temp directory under the user provided filename, we do this
# because pkginfo validates that the filename has a valid extension before
# reading it
with tempfile.NamedTemporaryFile('wb', dir=".", suffix=filename) as temp_file:
artifact_file = storage.open(artifact.file.name)
shutil.copyfileobj(artifact_file, temp_file)
temp_file.flush()
metadata = DIST_TYPES[packagetype](temp_file.name)
metadata.packagetype = packagetype
return metadata
def python_content_to_json(base_path, content_query, version=None):
"""
Converts a QuerySet of PythonPackageContent into the PyPi JSON format
https://www.python.org/dev/peps/pep-0566/
JSON metadata has:
info: Dict
last_serial: int
releases: Dict
urls: Dict
Returns None if version is specified but not found within content_query
"""
full_metadata = {"last_serial": 0} # For now the serial field isn't supported by Pulp
latest_content = latest_content_version(content_query, version)
if not latest_content:
return None
full_metadata.update({"info": python_content_to_info(latest_content[0])})
full_metadata.update({"releases": python_content_to_releases(content_query, base_path)})
full_metadata.update({"urls": python_content_to_urls(latest_content, base_path)})
return full_metadata
def latest_content_version(content_query, version):
"""
Walks through the content QuerySet and finds the instances that is the latest version.
If 'version' is specified, the function instead tries to find content instances
with that version and will return an empty list if nothing is found
"""
latest_version = version
latest_content = []
for content in content_query:
if version and parse(version) == parse(content.version):
latest_content.append(content)
elif not latest_version or parse(content.version) > parse(latest_version):
latest_content = [content]
latest_version = content.version
elif parse(content.version) == parse(latest_version):
latest_content.append(content)
return latest_content
def json_to_dict(data):
"""
Converts a JSON string into a Python dictionary.
Args:
data (string): JSON string
Returns:
dictionary: of JSON string
"""
if isinstance(data, dict):
return data
return json.loads(data)
def python_content_to_info(content):
"""
Takes in a PythonPackageContent instance and returns a dictionary of the Info fields
"""
return {
"name": content.name,
"version": content.version,
"summary": content.summary or "",
"keywords": content.keywords or "",
"description": content.description or "",
"description_content_type": content.description_content_type or "",
"bugtrack_url": None, # These two are basically never used
"docs_url": None,
"downloads": {"last_day": -1, "last_month": -1, "last_week": -1},
"download_url": content.download_url or "",
"home_page": content.home_page or "",
"author": content.author or "",
"author_email": content.author_email or "",
"maintainer": content.maintainer or "",
"maintainer_email": content.maintainer_email or "",
"license": content.license or "",
"requires_python": content.requires_python or None,
"package_url": content.project_url or "", # These two are usually identical
"project_url": content.project_url or "", # They also usually point to PyPI
"release_url": f"{content.project_url}{content.version}/" if content.project_url else "",
"project_urls": json_to_dict(content.project_urls) or None,
"platform": content.platform or "",
"requires_dist": json_to_dict(content.requires_dist) or None,
"classifiers": json_to_dict(content.classifiers) or None,
"yanked": False, # These are no longer used on PyPI, but are still present
"yanked_reason": None,
}
def python_content_to_releases(content_query, base_path):
"""
Takes a QuerySet of PythonPackageContent and returns a dictionary of releases
with each key being a version and value being a list of content for that version of the package
"""
releases = defaultdict(lambda: [])
for content in content_query:
releases[content.version].append(python_content_to_download_info(content, base_path))
return releases
def python_content_to_urls(contents, base_path):
"""
Takes the latest content in contents and returns a list of download information
"""
return [python_content_to_download_info(content, base_path) for content in contents]
def python_content_to_download_info(content, base_path):
"""
Takes in a PythonPackageContent and base path of the distribution to create a dictionary of
download information for that content. This dictionary is used by Releases and Urls.
"""
def find_artifact():
_art = content_artifact.artifact
if not _art:
from pulpcore.plugin import models
_art = models.RemoteArtifact.objects.filter(content_artifact=content_artifact).first()
return _art
content_artifact = content.contentartifact_set.first()
artifact = find_artifact()
origin = settings.CONTENT_ORIGIN.strip("/")
prefix = settings.CONTENT_PATH_PREFIX.strip("/")
base_path = base_path.strip("/")
url = "/".join((origin, prefix, base_path, content.filename))
return {
"comment_text": "",
"digests": {"md5": artifact.md5, "sha256": artifact.sha256},
"downloads": -1,
"filename": content.filename,
"has_sig": False,
"md5_digest": artifact.md5,
"packagetype": content.packagetype,
"python_version": content.python_version,
"requires_python": content.requires_python or None,
"size": artifact.size,
"upload_time": str(artifact.pulp_created),
"upload_time_iso_8601": str(artifact.pulp_created.isoformat()),
"url": url,
"yanked": False,
"yanked_reason": None
}
def write_simple_index(project_names, streamed=False):
"""Writes the simple index."""
simple = Template(simple_index_template)
context = {"projects": ((x, canonicalize_name(x)) for x in project_names)}
return simple.stream(**context) if streamed else simple.render(**context)
def write_simple_detail(project_name, project_packages, streamed=False):
"""Writes the simple detail page of a package."""
detail = Template(simple_detail_template)
context = {"project_name": project_name, "project_packages": project_packages}
return detail.stream(**context) if streamed else detail.render(**context)