Files
galaxy/scripts/extract_tool_sections_from_api.py
T
paperbenni bec3d3ad9c Drop support for Python 3.8
Drop Python 3.8 support in 5 Pulsar-compatible packages
(job_metrics, tool_util, tool_util_models, util, objectstore)
and run pyupgrade --py310-plus on their source code.

- Bump requires-python from >=3.8 to >=3.10 in all 5 packages
- Remove Python 3.8 and 3.9 classifiers
- Remove ruff per-file-ignores for UP rules on these paths
- Remove backports.zoneinfo conditional dependency
- Remove pydyf<0.11 pin from conditional-requirements.txt
- Update Makefile pyupgrade target (remove PY38_PYUPGRADE_PATHS)
- Update CI workflow to test with Python 3.10 instead of 3.8

Clean up unused deprecated typing imports after pyupgrade

Remove now-unused typing imports (Dict, List, Optional, Set, Tuple,
Type, Union) that became dead after pyupgrade --py310-plus converted
annotations to use built-in types and | syntax.

Also run ruff check --fix --select=UP007,UP045 across the entire
codebase to convert remaining Optional[X] -> X | None and
Union[X, Y] -> X | Y patterns.

Enable ruff UP007/UP045 for Python 3.10 union syntax

Remove UP007 (Union[X,Y] -> X | Y) and UP045 (Optional[X] -> X | None)
from the ruff ignore list and convert all type aliases across the
codebase. These rules were deferred while Python 3.9 was supported;
requires-python is now >=3.10.

A custom script was used because neither ruff --fix nor
pyupgrade --py310-plus converts Optional[X]/Union[X,Y] in type alias
positions (e.g. X = Union[A, B]) — they only handle annotation
positions (e.g. def f(x: Optional[int])). All 167 violations were
module-level type aliases. A few edge cases were fixed manually:
single-element Union[X,], typing.Union qualified refs, runtime
Optional[type] calls, and Annotated[Optional[...]] pydantic fields.

Fix UP007 autofix regression with string forward reference type aliases

Commit fa6bd955a0 enabled ruff UP007/UP045 and auto-fixed
module-level type aliases using Union with string forward
references, producing invalid 'str | str' expressions.

This was a known ruff bug (charliermarsh/ruff#826) that has since
been fixed in later ruff versions, but this codebase was
converted before the fix was in place.

Revert to Union syntax and restore TYPE_CHECKING imports that
ruff's TCH rule cleaned up as a side effect when it thought the
forward references were unused.
2026-06-25 15:12:05 +01:00

302 lines
11 KiB
Python

#!/usr/bin/env python3
"""
Refresh section-derived tool tags from a Galaxy server's `/api/tools` payload.
Defaults to the EU Galaxy instance and writes the curated tag map to
``config/tool_tag_mappings.yml`` next to this repo; both can be overridden
via CLI flags so the script works from any cwd.
"""
import argparse
import json
import re
import sys
import urllib.error
import urllib.request
from pathlib import Path
from typing import (
Any,
)
import yaml
from yaml.events import (
AliasEvent,
CollectionStartEvent,
NodeEvent,
ScalarEvent,
)
DEFAULT_API_URL = "https://usegalaxy.eu/api/tools/"
DEFAULT_OUTPUT_FILE = Path(__file__).resolve().parent.parent / "config" / "tool_tag_mappings.yml"
DEFAULT_TIMEOUT_SECONDS = 60
# Module-level globals overwritten from CLI args in ``main`` so the helper
# functions (which the test suite imports) keep their existing signatures.
API_URL = DEFAULT_API_URL
OUTPUT_FILE = DEFAULT_OUTPUT_FILE
SAFE_UNQUOTED_RE = re.compile(r"^[A-Za-z0-9_.-]+$")
YAML_BOOLEAN_LIKE_VALUES = {"", "~", "null", "Null", "NULL", "true", "True", "TRUE", "false", "False", "FALSE"}
class QuotedString(str):
"""Marker type for YAML strings that must be emitted with quotes."""
class QuotingDumper(yaml.SafeDumper):
"""YAML dumper that keeps scalar quoting explicit and predictable.
PyYAML's default dumper changes quoting heuristically when scalar lengths
cross internal thresholds, which churns the output for diff review every
time we regenerate the file. A round-trip dumper from ``ruamel.yaml`` would
avoid this, but ruamel.yaml isn't a Galaxy runtime dependency and this
script is admin-only, so we patch the simple-key heuristic instead.
"""
def check_simple_key(self):
length = 0
if isinstance(self.event, NodeEvent) and self.event.anchor is not None:
if self.prepared_anchor is None:
self.prepared_anchor = self.prepare_anchor(self.event.anchor)
length += len(self.prepared_anchor)
if isinstance(self.event, (ScalarEvent, CollectionStartEvent)) and self.event.tag is not None:
if self.prepared_tag is None:
self.prepared_tag = self.prepare_tag(self.event.tag)
length += len(self.prepared_tag)
if isinstance(self.event, ScalarEvent):
if self.analysis is None:
self.analysis = self.analyze_scalar(self.event.value)
length += len(self.analysis.scalar)
return length < 4096 and (
isinstance(self.event, AliasEvent)
or (isinstance(self.event, ScalarEvent) and not self.analysis.empty and not self.analysis.multiline)
or self.check_empty_sequence()
or self.check_empty_mapping()
)
def represent_quoted_string(dumper: yaml.Dumper, data: str) -> yaml.ScalarNode:
return dumper.represent_scalar("tag:yaml.org,2002:str", data, style='"')
QuotingDumper.add_representer(QuotedString, represent_quoted_string)
def fetch_tools(api_url: str | None = None, timeout: float = DEFAULT_TIMEOUT_SECONDS) -> list[dict[str, Any]]:
"""Fetch tools from a Galaxy `/api/tools` endpoint.
Raises ``RuntimeError`` on transport / HTTP / decoding failures so the caller
can fail fast instead of silently overwriting the YAML with a no-op result.
"""
target_url = api_url or API_URL
print(f"Fetching tools from {target_url}...")
try:
with urllib.request.urlopen(target_url, timeout=timeout) as response:
return json.loads(response.read().decode("utf-8"))
except (urllib.error.URLError, TimeoutError) as exc:
raise RuntimeError(f"Could not reach {target_url}: {exc}") from exc
except (json.JSONDecodeError, UnicodeDecodeError) as exc:
raise RuntimeError(f"Server at {target_url} returned an unparseable response: {exc}") from exc
def normalize_section_alias(value: str) -> str:
return re.sub(r"_+", "_", re.sub(r"[^a-z0-9]+", "_", value.lower())).strip("_")
def tool_id_candidates(tool_id: str) -> list[str]:
candidates = [tool_id]
if tool_id.startswith("toolshed.") and tool_id.count("/") >= 5:
short_id = tool_id.rsplit("/", 1)[0]
if short_id not in candidates:
candidates.append(short_id)
return candidates
def extract_sections_and_tools(data: list[dict[str, Any]]) -> tuple[dict[str, list[str]], dict[str, str]]:
"""Extract ToolSections and their tools from API response."""
sections: dict[str, list[str]] = {}
section_ids: dict[str, str] = {}
for item in data:
if item.get("model_class") == "ToolSection":
section_id = item.get("id")
section_name = item.get("name")
if section_id and section_name:
section_ids[section_id] = section_name
print(f"Found section: {section_name} (id: {section_id})")
# Extract tools in this section
tools = []
for elem in item.get("elems", []):
model_class = elem.get("model_class", "")
if isinstance(model_class, str) and model_class.endswith("Tool"):
tool_id = elem.get("id")
if tool_id:
for candidate in tool_id_candidates(tool_id):
if candidate not in tools:
tools.append(candidate)
if tools:
sections.setdefault(section_name, []).extend(tools)
print(f" - Contains {len(tools)} tools")
return sections, section_ids
def load_existing_mappings() -> dict[str, Any]:
"""Load existing tool_tag_mappings.yml"""
if OUTPUT_FILE.exists():
with open(OUTPUT_FILE, encoding="utf-8") as f:
return yaml.safe_load(f)
return {"tool_tags": {}}
def normalize_tool_tags(tool_tags: dict[str, list[str]]) -> dict[str, list[str]]:
normalized: dict[str, list[str]] = {}
for tool_id, tags in tool_tags.items():
unique_tags: list[str] = []
for tag in tags:
if tag not in unique_tags:
unique_tags.append(tag)
normalized[tool_id] = unique_tags
return normalized
def validate_mappings_data(data: dict[str, Any]) -> dict[str, dict[str, list[str]]]:
tool_tags = data.get("tool_tags", {})
if not isinstance(tool_tags, dict):
raise ValueError("tool_tag_mappings.yml must contain a top-level 'tool_tags' mapping.")
for tool_id, tags in tool_tags.items():
if not isinstance(tool_id, str):
raise ValueError(f"Tool id {tool_id!r} must be a string.")
if not isinstance(tags, list) or not all(isinstance(tag, str) for tag in tags):
raise ValueError(f"Tags for tool {tool_id!r} must be a list of strings.")
return {"tool_tags": normalize_tool_tags(tool_tags)}
def needs_quotes(value: str) -> bool:
return value in YAML_BOOLEAN_LIKE_VALUES or not SAFE_UNQUOTED_RE.fullmatch(value)
def prepare_for_dump(value: Any) -> Any:
if isinstance(value, dict):
return {prepare_for_dump(key): prepare_for_dump(item) for key, item in value.items()}
if isinstance(value, list):
return [prepare_for_dump(item) for item in value]
if isinstance(value, str) and needs_quotes(value):
return QuotedString(value)
return value
def update_mappings(
existing_data: dict[str, Any],
sections: dict[str, list[str]],
section_ids: dict[str, str],
) -> dict[str, dict[str, list[str]]]:
"""Refresh section tags while preserving existing non-section curated tags."""
section_aliases = {normalize_section_alias(alias) for alias in set(section_ids) | set(section_ids.values())}
tool_tags: dict[str, list[str]] = {}
for tool_id, tags in existing_data.get("tool_tags", {}).items():
custom_tags = [tag for tag in tags if normalize_section_alias(tag) not in section_aliases]
if custom_tags:
tool_tags[tool_id] = custom_tags
# Track statistics
new_mappings = 0
updated_mappings = 0
for section_name, tools in sections.items():
tag = section_name
for tool_id in tools:
if tool_id in tool_tags:
# Check if tag already exists
if tag not in tool_tags[tool_id]:
tool_tags[tool_id].append(tag)
updated_mappings += 1
print(f" Updated {tool_id} with tag: {tag}")
else:
tool_tags[tool_id] = [tag]
new_mappings += 1
print(f" Added {tool_id} with tag: {tag}")
print("\nSummary:")
print(f" - New tool mappings: {new_mappings}")
print(f" - Updated tool mappings: {updated_mappings}")
print(f" - Total tools with tags: {len(tool_tags)}")
return {"tool_tags": normalize_tool_tags(tool_tags)}
def save_mappings(data: dict[str, Any]) -> None:
"""Save updated tool_tag_mappings.yml"""
validated_data = validate_mappings_data(data)
serialized = yaml.dump(
prepare_for_dump(validated_data),
Dumper=QuotingDumper,
default_flow_style=False,
sort_keys=False,
allow_unicode=True,
width=1000,
)
# Validate that the serialized YAML round-trips cleanly.
round_trip = yaml.safe_load(serialized)
if round_trip != validated_data:
raise ValueError("Serialized YAML did not round-trip correctly.")
print(f"\nSaving to {OUTPUT_FILE}...")
with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
f.write(serialized)
print("Done!")
def _build_arg_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--api-url",
default=DEFAULT_API_URL,
help=f"Galaxy /api/tools endpoint to scrape (default: {DEFAULT_API_URL}).",
)
parser.add_argument(
"--output",
type=Path,
default=DEFAULT_OUTPUT_FILE,
help=f"Path of the curated tool_tag_mappings.yml to write (default: {DEFAULT_OUTPUT_FILE}).",
)
parser.add_argument(
"--timeout",
type=float,
default=DEFAULT_TIMEOUT_SECONDS,
help=f"HTTP timeout in seconds (default: {DEFAULT_TIMEOUT_SECONDS}).",
)
return parser
def main(argv: list[str] | None = None) -> int:
args = _build_arg_parser().parse_args(argv)
global API_URL, OUTPUT_FILE
API_URL = args.api_url
OUTPUT_FILE = args.output
try:
data = fetch_tools(api_url=args.api_url, timeout=args.timeout)
except RuntimeError as exc:
print(f"Error: {exc}", file=sys.stderr)
return 2
sections, section_ids = extract_sections_and_tools(data)
if not sections:
print("No sections found!", file=sys.stderr)
return 1
existing_data = load_existing_mappings()
updated_data = update_mappings(existing_data, sections, section_ids)
save_mappings(updated_data)
return 0
if __name__ == "__main__":
sys.exit(main())