/
/
1#!/usr/bin/env python3
2"""
3Check PyPI package metadata for security and supply chain concerns.
4
5This script checks new or updated Python dependencies for suspicious indicators
6that might suggest supply chain attacks or unmaintained packages.
7"""
8
9# ruff: noqa: T201, S310, RUF001, PLR0915
10import json
11import re
12import sys
13import urllib.request
14from datetime import datetime
15from typing import Any
16
17# OSI-approved and common compatible licenses
18COMPATIBLE_LICENSES = {
19 "MIT",
20 "Apache-2.0",
21 "Apache Software License",
22 "BSD",
23 "BSD-3-Clause",
24 "BSD-2-Clause",
25 "ISC",
26 "Python Software Foundation License",
27 "PSF",
28 "LGPL",
29 "MPL-2.0",
30 "Unlicense",
31 "CC0",
32 # compact spellings that would not survive the boundary check as a suffix of the name above
33 "PSFL",
34 "ISCL",
35 "LGPLv2",
36 "LGPLv3",
37}
38
39# Licenses recognised only as the whole value, because their wording also turns up in prose that
40# says the opposite ("this software is not in the public domain")
41EXACT_LICENSES = {
42 "Public Domain",
43}
44
45# SPDX identifiers accepted in a PEP 639 `license_expression`
46COMPATIBLE_SPDX_LICENSES = {
47 "0BSD",
48 "APACHE-2.0",
49 "BSD-2-CLAUSE",
50 "BSD-3-CLAUSE",
51 "BSL-1.0",
52 "CC0-1.0",
53 "ISC",
54 "LGPL-2.0",
55 "LGPL-2.0-ONLY",
56 "LGPL-2.0-OR-LATER",
57 "LGPL-2.1",
58 "LGPL-2.1-ONLY",
59 "LGPL-2.1-OR-LATER",
60 "LGPL-3.0",
61 "LGPL-3.0-ONLY",
62 "LGPL-3.0-OR-LATER",
63 "MIT",
64 "MIT-0",
65 "MIT-CMU",
66 "MPL-2.0",
67 "PSF-2.0",
68 "PYTHON-2.0",
69 "UNLICENSE",
70 "ZLIB",
71}
72
73# License families that are incompatible with the project (LGPL excluded)
74PROBLEMATIC_LICENSES = ("GPL", "AGPL", "SSPL")
75
76# Deepest group nesting accepted in an SPDX expression
77MAX_SPDX_NESTING = 10
78
79# Common packages to check for typosquatting (popular Python packages)
80POPULAR_PACKAGES = {
81 "requests",
82 "urllib3",
83 "setuptools",
84 "certifi",
85 "pip",
86 "numpy",
87 "pandas",
88 "boto3",
89 "botocore",
90 "awscli",
91 "django",
92 "flask",
93 "sqlalchemy",
94 "pytest",
95 "pydantic",
96 "aiohttp",
97 "fastapi",
98}
99
100
101def check_typosquatting(package_name: str) -> str | None:
102 """
103 Check if package name might be typosquatting a popular package.
104
105 :param package_name: The package name to check.
106 """
107 package_lower = package_name.lower().replace("-", "").replace("_", "")
108
109 for popular in POPULAR_PACKAGES:
110 popular_normalized = popular.lower().replace("-", "").replace("_", "")
111
112 # Check for common typosquatting techniques
113 if package_lower == popular_normalized:
114 continue # Exact match is fine
115
116 # Check edit distance (1-2 character changes)
117 if len(package_lower) == len(popular_normalized):
118 differences = sum(
119 c1 != c2 for c1, c2 in zip(package_lower, popular_normalized, strict=True)
120 )
121 if differences == 1:
122 return f"Suspicious: Very similar to popular package '{popular}'"
123
124 # Check for common substitutions
125 substitutions = [
126 ("0", "o"),
127 ("1", "l"),
128 ("1", "i"),
129 ]
130 for old, new in substitutions:
131 if old in package_lower:
132 test_name = package_lower.replace(old, new)
133 if test_name == popular_normalized:
134 return f"Suspicious: Character substitution of popular package '{popular}'"
135
136 return None
137
138
139def get_package_license(info: dict[str, Any]) -> tuple[str, bool]:
140 """
141 Resolve the license of a package from its PyPI metadata.
142
143 Returns the license and whether it is a PEP 639 SPDX expression.
144
145 :param info: The `info` section of the PyPI JSON response.
146 """
147 # packages that adopted PEP 639 declare an SPDX expression, which is more precise than the
148 # free-form field and the classifiers, and is usually the only license metadata they carry
149 if license_expression := (info.get("license_expression") or "").strip():
150 return license_expression, True
151
152 if license_str := (info.get("license") or "").strip():
153 return license_str, False
154
155 license_classifiers = []
156 for classifier in info.get("classifiers") or []:
157 # the license itself is the last segment, e.g. "License :: OSI Approved :: MIT License",
158 # except for "License :: OSI Approved", which names no license at all
159 parts = [part.strip() for part in classifier.split("::")]
160 if parts[0] == "License" and len(parts) > 1 and parts[-1] != "OSI Approved":
161 license_classifiers.append(parts[-1])
162
163 if license_classifiers:
164 # classifiers do not say how they relate to each other, so report the one that fails the
165 # compatibility check rather than letting a permissive entry hide a copyleft one
166 return next(
167 (name for name in license_classifiers if not check_license_compatibility(name)[0]),
168 license_classifiers[0],
169 ), False
170
171 return "Unknown", False
172
173
174def check_license_compatibility(
175 license_str: str, spdx_expression: bool = False
176) -> tuple[bool, str]:
177 """
178 Check if license is compatible with the project.
179
180 :param license_str: The license string from PyPI.
181 :param spdx_expression: Whether the string is a PEP 639 SPDX expression.
182 """
183 if not license_str or license_str == "Unknown":
184 return False, "No license information"
185
186 license_upper = license_str.upper()
187 spdx_compatible, joins_licenses = _evaluate_spdx_expression(license_str)
188 # a value that reads as an expression joining licenses is precise enough to judge on the
189 # names in it alone, whichever field it came from
190 spdx_expression = spdx_expression or joins_licenses
191
192 if spdx_compatible:
193 return True, f"Compatible ({license_str})"
194
195 copyleft = _is_copyleft(license_upper)
196
197 # whatever the evaluator did not accept in an expression names a license that is not on the
198 # allow list, so guessing from the wording would only weaken the check. For free-form fields
199 # the wording is all we have, but a name we do recognise there must not end up approving a
200 # copyleft or custom license next to it
201 guessable = not spdx_expression and not copyleft and "LICENSEREF" not in license_upper
202 if spdx_compatible is None and guessable:
203 for compatible in COMPATIBLE_LICENSES | EXACT_LICENSES:
204 if _names_license(license_upper, compatible):
205 return True, f"Compatible ({license_str})"
206
207 if copyleft:
208 return False, f"Incompatible copyleft license ({license_str})"
209
210 # Unknown license
211 return False, f"Unknown/unverified license ({license_str})"
212
213
214def parse_requirement(line: str) -> str | None:
215 """
216 Extract package name from a requirement line.
217
218 :param line: A line from requirements.txt (e.g., "package==1.0.0" or "package>=1.0")
219 """
220 line = line.strip()
221 if not line or line.startswith("#"):
222 return None
223
224 # Handle various requirement formats
225 # package==1.0.0, package>=1.0, package[extra]>=1.0, etc.
226 match = re.match(r"^([a-zA-Z0-9_-]+)", line)
227 if match:
228 return match.group(1).lower()
229 return None
230
231
232def get_pypi_metadata(package_name: str) -> dict[str, Any] | None:
233 """
234 Fetch package metadata from PyPI JSON API.
235
236 :param package_name: The name of the package to check.
237 """
238 url = f"https://pypi.org/pypi/{package_name}/json"
239
240 try:
241 with urllib.request.urlopen(url, timeout=10) as response:
242 return json.loads(response.read())
243 except urllib.error.HTTPError as err:
244 if err.code == 404:
245 print(f"â Package '{package_name}' not found on PyPI")
246 else:
247 print(f"â ï¸ Error fetching metadata for '{package_name}': {err}")
248 return None
249 except Exception as err:
250 print(f"â ï¸ Error fetching metadata for '{package_name}': {err}")
251 return None
252
253
254def check_package(package_name: str) -> dict[str, Any]:
255 """
256 Check a single package for security concerns.
257
258 :param package_name: The name of the package to check.
259 """
260 data = get_pypi_metadata(package_name)
261
262 if not data:
263 return {
264 "name": package_name,
265 "error": "Could not fetch package metadata",
266 "risk_level": "unknown",
267 "warnings": [],
268 }
269
270 info = data.get("info", {})
271 releases = data.get("releases", {})
272
273 # Get package age
274 upload_times = []
275 for release_files in releases.values():
276 if release_files:
277 for file_info in release_files:
278 if "upload_time" in file_info:
279 try:
280 upload_time_str = file_info["upload_time"]
281 # Handle both formats: with 'Z' suffix or with timezone
282 if upload_time_str.endswith("Z"):
283 upload_time_str = upload_time_str[:-1] + "+00:00"
284 upload_time = datetime.fromisoformat(upload_time_str)
285 upload_times.append(upload_time)
286 except ValueError, AttributeError:
287 continue
288
289 first_upload = min(upload_times) if upload_times else None
290 age_days = (datetime.now(first_upload.tzinfo) - first_upload).days if first_upload else 0
291
292 # Extract metadata
293 project_urls = info.get("project_urls") or {}
294 homepage = info.get("home_page") or project_urls.get("Homepage")
295 source = project_urls.get("Source") or project_urls.get("Repository")
296
297 # Run automated security checks
298 typosquat_check = check_typosquatting(package_name)
299 package_license, spdx_expression = get_package_license(info)
300 license_compatible, license_status = check_license_compatibility(
301 package_license, spdx_expression
302 )
303
304 checks = {
305 "name": package_name,
306 "version": info.get("version", "unknown"),
307 "age_days": age_days,
308 "total_releases": len(releases),
309 "has_homepage": bool(homepage),
310 "has_source": bool(source),
311 "author": info.get("author") or info.get("maintainer") or "Unknown",
312 "license": package_license,
313 "summary": info.get("summary", "No description"),
314 "warnings": [],
315 "info_items": [],
316 "risk_level": "low",
317 "automated_checks": {
318 "trusted_source": bool(source),
319 "typosquatting": typosquat_check is None,
320 "license_compatible": license_compatible,
321 },
322 "check_details": {
323 "typosquatting": typosquat_check or "â No typosquatting detected",
324 "license": license_status,
325 },
326 }
327
328 # Check for suspicious indicators
329 risk_score = 0
330
331 # Typosquatting check
332 if typosquat_check:
333 checks["warnings"].append(typosquat_check)
334 risk_score += 5 # High risk
335
336 # License check
337 if not license_compatible:
338 checks["warnings"].append(f"License issue: {license_status}")
339 risk_score += 2
340
341 if age_days < 30:
342 checks["warnings"].append(f"Very new package (only {age_days} days old)")
343 risk_score += 3
344 elif age_days < 90:
345 checks["warnings"].append(f"Relatively new package ({age_days} days old)")
346 risk_score += 1
347
348 if checks["total_releases"] < 3:
349 checks["warnings"].append(f"Very few releases (only {checks['total_releases']})")
350 risk_score += 2
351
352 if not source:
353 checks["warnings"].append("No source repository linked")
354 risk_score += 2
355
356 if not homepage and not source:
357 checks["warnings"].append("No homepage or source repository")
358 risk_score += 1
359
360 if checks["author"] == "Unknown":
361 checks["warnings"].append("No author information available")
362 risk_score += 1
363
364 # Add informational items
365 checks["info_items"].append(f"Age: {age_days} days")
366 checks["info_items"].append(f"Releases: {checks['total_releases']}")
367 checks["info_items"].append(f"Author: {checks['author']}")
368 checks["info_items"].append(f"License: {checks['license']}")
369 if source:
370 checks["info_items"].append(f"Source: {source}")
371
372 # Determine risk level
373 if risk_score >= 5:
374 checks["risk_level"] = "high"
375 elif risk_score >= 3:
376 checks["risk_level"] = "medium"
377 else:
378 checks["risk_level"] = "low"
379
380 return checks
381
382
383def format_check_result(result: dict[str, Any]) -> str:
384 """
385 Format a check result for display.
386
387 :param result: The check result dictionary.
388 """
389 risk_emoji = {"high": "ð´", "medium": "ð¡", "low": "ð¢", "unknown": "âª"}
390 version = result.get("version", "unknown")
391
392 lines = [f"\n{risk_emoji[result['risk_level']]} **{result['name']}** (v{version})"]
393
394 if result.get("error"):
395 lines.append(f" â {result['error']}")
396 return "\n".join(lines)
397
398 if result.get("summary"):
399 lines.append(f" ð {result['summary']}")
400
401 if result.get("info_items"):
402 for item in result["info_items"]:
403 lines.append(f" â¹ï¸ {item}")
404
405 if result.get("warnings"):
406 for warning in result["warnings"]:
407 lines.append(f" â ï¸ {warning}")
408
409 return "\n".join(lines)
410
411
412def main() -> int:
413 """Run the package safety check."""
414 if len(sys.argv) < 2:
415 print("Usage: check_package_safety.py <requirements_file_or_package_name>")
416 print(" Or: check_package_safety.py package1 package2 package3")
417 return 1
418
419 packages = []
420
421 # Check if first argument is a file
422 if len(sys.argv) == 2 and sys.argv[1].endswith(".txt"):
423 try:
424 with open(sys.argv[1]) as f:
425 for line in f:
426 package = parse_requirement(line)
427 if package:
428 packages.append(package)
429 except FileNotFoundError:
430 print(f"Error: File '{sys.argv[1]}' not found")
431 return 1
432 else:
433 # Treat arguments as package names
434 packages = [arg.lower() for arg in sys.argv[1:]]
435
436 if not packages:
437 print("No packages to check")
438 return 0
439
440 print(f"Checking {len(packages)} package(s)...\n")
441 print("=" * 80)
442
443 results = []
444 for package in packages:
445 result = check_package(package)
446 results.append(result)
447 print(format_check_result(result))
448
449 print("\n" + "=" * 80)
450
451 # Automated checks summary
452 all_trusted = all(r.get("automated_checks", {}).get("trusted_source", False) for r in results)
453 all_no_typosquat = all(
454 r.get("automated_checks", {}).get("typosquatting", False) for r in results
455 )
456 all_license_ok = all(
457 r.get("automated_checks", {}).get("license_compatible", False) for r in results
458 )
459
460 print("\nð¤ Automated Security Checks:")
461 trusted_msg = (
462 "All packages have source repositories"
463 if all_trusted
464 else "Some packages missing source info"
465 )
466 print(f" {'â
' if all_trusted else 'â'} Trusted Sources: {trusted_msg}")
467
468 typosquat_msg = (
469 "No suspicious package names detected"
470 if all_no_typosquat
471 else "Possible typosquatting detected!"
472 )
473 print(f" {'â
' if all_no_typosquat else 'â'} Typosquatting: {typosquat_msg}")
474
475 license_msg = (
476 "All licenses are compatible" if all_license_ok else "Some license issues detected"
477 )
478 print(f" {'â
' if all_license_ok else 'â ï¸ '} License Compatibility: {license_msg}")
479
480 # Summary
481 high_risk = sum(1 for r in results if r["risk_level"] == "high")
482 medium_risk = sum(1 for r in results if r["risk_level"] == "medium")
483 low_risk = sum(1 for r in results if r["risk_level"] == "low")
484
485 print(f"\nð Summary: {len(results)} packages checked")
486 if high_risk:
487 print(f" ð´ High risk: {high_risk}")
488 if medium_risk:
489 print(f" ð¡ Medium risk: {medium_risk}")
490 print(f" ð¢ Low risk: {low_risk}")
491
492 if high_risk > 0:
493 print("\nâ ï¸ High-risk packages detected! Manual review strongly recommended.")
494 return 2
495 if medium_risk > 0:
496 print("\nâ ï¸ Medium-risk packages detected. Please review before merging.")
497 return 1
498
499 print("\nâ
All packages passed basic safety checks.")
500 return 0
501
502
503class _SpdxSyntaxError(Exception):
504 """Raised when a string does not follow the SPDX expression grammar."""
505
506
507def _evaluate_spdx_expression(license_str: str) -> tuple[bool | None, bool]:
508 """
509 Check an SPDX license expression (PEP 639) against the allow list.
510
511 Returns whether the expression is compatible, which is None when it names a license that is
512 neither known-compatible nor known-problematic or when the string is not an expression at
513 all, together with whether the part that did read as one joined several licenses.
514
515 :param license_str: The license string to evaluate, e.g. "MIT OR Apache-2.0".
516 """
517 tokens = re.findall(r"\(|\)|[^\s()]+", license_str)
518 if not tokens:
519 return None, False
520 # real expressions nest a group or two at most; refuse anything deeper rather than recursing
521 # into it, and refuse outright as the fallback would match on a name nested inside
522 if _max_group_depth(tokens) > MAX_SPDX_NESTING:
523 return False, True
524
525 remaining = list(tokens)
526 try:
527 result = _evaluate_spdx_tokens(remaining)
528 if remaining:
529 # anything left over means we did not understand the string as a whole
530 raise _SpdxSyntaxError
531 except _SpdxSyntaxError:
532 result = None
533
534 # judge the shape on the tokens that were read as an expression, so that prose merely holding
535 # the word "and" is not mistaken for one, while a malformed expression still is
536 read = tokens[: len(tokens) - len(remaining)]
537 return result, any(token.upper() in ("AND", "OR", "WITH") for token in read)
538
539
540def _evaluate_spdx_tokens(tokens: list[str]) -> bool | None:
541 """
542 Evaluate the leading SPDX expression, consuming the tokens it covers.
543
544 Any alternative that is compatible makes the whole expression compatible.
545
546 :param tokens: The remaining tokens of the expression.
547 """
548 result = _evaluate_spdx_term(tokens)
549 while tokens and tokens[0].upper() == "OR":
550 tokens.pop(0)
551 term = _evaluate_spdx_term(tokens)
552 if True in (result, term):
553 result = True
554 elif None in (result, term):
555 result = None
556 return result
557
558
559def _evaluate_spdx_term(tokens: list[str]) -> bool | None:
560 """
561 Evaluate the leading "AND" sequence, which binds tighter than "OR" in SPDX.
562
563 Every operand of the sequence has to be compatible on its own.
564
565 :param tokens: The remaining tokens of the expression.
566 """
567 result = _evaluate_spdx_operand(tokens)
568 while tokens and tokens[0].upper() == "AND":
569 tokens.pop(0)
570 operand = _evaluate_spdx_operand(tokens)
571 if False in (result, operand):
572 result = False
573 elif None in (result, operand):
574 result = None
575 return result
576
577
578def _evaluate_spdx_operand(tokens: list[str]) -> bool | None:
579 """
580 Evaluate a single SPDX operand: a parenthesised expression or a license identifier.
581
582 :param tokens: The remaining tokens of the expression.
583 """
584 if not tokens or _is_spdx_operator(tokens[0]):
585 raise _SpdxSyntaxError
586
587 if tokens[0] == "(":
588 tokens.pop(0)
589 result = _evaluate_spdx_tokens(tokens)
590 if not tokens or tokens.pop(0) != ")":
591 raise _SpdxSyntaxError
592 return result
593
594 # a single trailing "+" is the deprecated "or later" marker and does not change the license
595 identifier = tokens.pop(0).upper().removesuffix("+")
596 # a "WITH <exception>" suffix only grants extra permissions, so the identifier decides
597 if tokens and tokens[0].upper() == "WITH":
598 tokens.pop(0)
599 if not tokens or _is_spdx_operator(tokens[0]):
600 raise _SpdxSyntaxError
601 # ...but a license we refuse is not an exception, whatever it is written behind
602 if _is_copyleft(tokens.pop(0).upper()):
603 return False
604
605 return True if identifier in COMPATIBLE_SPDX_LICENSES else None
606
607
608def _names_license(license_upper: str, name: str) -> bool:
609 """
610 Return whether an upper-cased license string names the given license.
611
612 :param license_upper: The upper-cased license string to search.
613 :param name: The license name to look for, e.g. "MPL-2.0".
614 """
615 # tolerate spelling variants of the separators, so that "MPL 2.0" is read as "MPL-2.0", but
616 # match whole words only, so the name is neither assembled out of unrelated ones ("this
617 # copyright" holding an "ISC") nor read out of one that merely starts the same ("MITigation")
618 if name in EXACT_LICENSES:
619 return license_upper.strip() == name.upper()
620
621 parts = [re.escape(part) for part in re.findall(r"[A-Z0-9]+", name.upper())]
622 return re.search(rf"\b{r'[^A-Z0-9]*'.join(parts)}(?![A-Z0-9])", license_upper) is not None
623
624
625def _is_spdx_operator(token: str) -> bool:
626 """
627 Return whether a token joins or closes expressions instead of naming a license.
628
629 :param token: The token to inspect.
630 """
631 return token == ")" or token.upper() in ("AND", "OR", "WITH")
632
633
634def _is_copyleft(license_upper: str) -> bool:
635 """
636 Return whether an upper-cased license string names a copyleft family we do not accept.
637
638 :param license_upper: The upper-cased license string to inspect.
639 """
640 # drop the LGPL mentions first, so that a GPL term next to an LGPL one is still spotted
641 without_lgpl = license_upper.replace("LGPL", "")
642 return any(problem in without_lgpl for problem in PROBLEMATIC_LICENSES)
643
644
645def _max_group_depth(tokens: list[str]) -> int:
646 """
647 Return how deeply the parentheses in an expression nest.
648
649 :param tokens: The tokens of the expression.
650 """
651 depth = 0
652 deepest = 0
653 for token in tokens:
654 if token == "(":
655 depth += 1
656 deepest = max(deepest, depth)
657 elif token == ")":
658 depth = max(0, depth - 1)
659 return deepest
660
661
662if __name__ == "__main__":
663 sys.exit(main())
664