mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-08-31 01:02:04 +08:00
bec3d3ad9c
Drop Python 3.8 support in 5 Pulsar-compatible packages (job_metrics, tool_util, tool_util_models, util, objectstore) and run pyupgrade --py310-plus on their source code. - Bump requires-python from >=3.8 to >=3.10 in all 5 packages - Remove Python 3.8 and 3.9 classifiers - Remove ruff per-file-ignores for UP rules on these paths - Remove backports.zoneinfo conditional dependency - Remove pydyf<0.11 pin from conditional-requirements.txt - Update Makefile pyupgrade target (remove PY38_PYUPGRADE_PATHS) - Update CI workflow to test with Python 3.10 instead of 3.8 Clean up unused deprecated typing imports after pyupgrade Remove now-unused typing imports (Dict, List, Optional, Set, Tuple, Type, Union) that became dead after pyupgrade --py310-plus converted annotations to use built-in types and | syntax. Also run ruff check --fix --select=UP007,UP045 across the entire codebase to convert remaining Optional[X] -> X | None and Union[X, Y] -> X | Y patterns. Enable ruff UP007/UP045 for Python 3.10 union syntax Remove UP007 (Union[X,Y] -> X | Y) and UP045 (Optional[X] -> X | None) from the ruff ignore list and convert all type aliases across the codebase. These rules were deferred while Python 3.9 was supported; requires-python is now >=3.10. A custom script was used because neither ruff --fix nor pyupgrade --py310-plus converts Optional[X]/Union[X,Y] in type alias positions (e.g. X = Union[A, B]) — they only handle annotation positions (e.g. def f(x: Optional[int])). All 167 violations were module-level type aliases. A few edge cases were fixed manually: single-element Union[X,], typing.Union qualified refs, runtime Optional[type] calls, and Annotated[Optional[...]] pydantic fields. Fix UP007 autofix regression with string forward reference type aliases Commit fa6bd955a0 enabled ruff UP007/UP045 and auto-fixed module-level type aliases using Union with string forward references, producing invalid 'str | str' expressions. This was a known ruff bug (charliermarsh/ruff#826) that has since been fixed in later ruff versions, but this codebase was converted before the fix was in place. Revert to Union syntax and restore TYPE_CHECKING imports that ruff's TCH rule cleaned up as a side effect when it thought the forward references were unused.
302 lines
11 KiB
Python
302 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Refresh section-derived tool tags from a Galaxy server's `/api/tools` payload.
|
|
|
|
Defaults to the EU Galaxy instance and writes the curated tag map to
|
|
``config/tool_tag_mappings.yml`` next to this repo; both can be overridden
|
|
via CLI flags so the script works from any cwd.
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import sys
|
|
import urllib.error
|
|
import urllib.request
|
|
from pathlib import Path
|
|
from typing import (
|
|
Any,
|
|
)
|
|
|
|
import yaml
|
|
from yaml.events import (
|
|
AliasEvent,
|
|
CollectionStartEvent,
|
|
NodeEvent,
|
|
ScalarEvent,
|
|
)
|
|
|
|
DEFAULT_API_URL = "https://usegalaxy.eu/api/tools/"
|
|
DEFAULT_OUTPUT_FILE = Path(__file__).resolve().parent.parent / "config" / "tool_tag_mappings.yml"
|
|
DEFAULT_TIMEOUT_SECONDS = 60
|
|
|
|
# Module-level globals overwritten from CLI args in ``main`` so the helper
|
|
# functions (which the test suite imports) keep their existing signatures.
|
|
API_URL = DEFAULT_API_URL
|
|
OUTPUT_FILE = DEFAULT_OUTPUT_FILE
|
|
SAFE_UNQUOTED_RE = re.compile(r"^[A-Za-z0-9_.-]+$")
|
|
YAML_BOOLEAN_LIKE_VALUES = {"", "~", "null", "Null", "NULL", "true", "True", "TRUE", "false", "False", "FALSE"}
|
|
|
|
|
|
class QuotedString(str):
|
|
"""Marker type for YAML strings that must be emitted with quotes."""
|
|
|
|
|
|
class QuotingDumper(yaml.SafeDumper):
|
|
"""YAML dumper that keeps scalar quoting explicit and predictable.
|
|
|
|
PyYAML's default dumper changes quoting heuristically when scalar lengths
|
|
cross internal thresholds, which churns the output for diff review every
|
|
time we regenerate the file. A round-trip dumper from ``ruamel.yaml`` would
|
|
avoid this, but ruamel.yaml isn't a Galaxy runtime dependency and this
|
|
script is admin-only, so we patch the simple-key heuristic instead.
|
|
"""
|
|
|
|
def check_simple_key(self):
|
|
length = 0
|
|
if isinstance(self.event, NodeEvent) and self.event.anchor is not None:
|
|
if self.prepared_anchor is None:
|
|
self.prepared_anchor = self.prepare_anchor(self.event.anchor)
|
|
length += len(self.prepared_anchor)
|
|
if isinstance(self.event, (ScalarEvent, CollectionStartEvent)) and self.event.tag is not None:
|
|
if self.prepared_tag is None:
|
|
self.prepared_tag = self.prepare_tag(self.event.tag)
|
|
length += len(self.prepared_tag)
|
|
if isinstance(self.event, ScalarEvent):
|
|
if self.analysis is None:
|
|
self.analysis = self.analyze_scalar(self.event.value)
|
|
length += len(self.analysis.scalar)
|
|
return length < 4096 and (
|
|
isinstance(self.event, AliasEvent)
|
|
or (isinstance(self.event, ScalarEvent) and not self.analysis.empty and not self.analysis.multiline)
|
|
or self.check_empty_sequence()
|
|
or self.check_empty_mapping()
|
|
)
|
|
|
|
|
|
def represent_quoted_string(dumper: yaml.Dumper, data: str) -> yaml.ScalarNode:
|
|
return dumper.represent_scalar("tag:yaml.org,2002:str", data, style='"')
|
|
|
|
|
|
QuotingDumper.add_representer(QuotedString, represent_quoted_string)
|
|
|
|
|
|
def fetch_tools(api_url: str | None = None, timeout: float = DEFAULT_TIMEOUT_SECONDS) -> list[dict[str, Any]]:
|
|
"""Fetch tools from a Galaxy `/api/tools` endpoint.
|
|
|
|
Raises ``RuntimeError`` on transport / HTTP / decoding failures so the caller
|
|
can fail fast instead of silently overwriting the YAML with a no-op result.
|
|
"""
|
|
target_url = api_url or API_URL
|
|
print(f"Fetching tools from {target_url}...")
|
|
try:
|
|
with urllib.request.urlopen(target_url, timeout=timeout) as response:
|
|
return json.loads(response.read().decode("utf-8"))
|
|
except (urllib.error.URLError, TimeoutError) as exc:
|
|
raise RuntimeError(f"Could not reach {target_url}: {exc}") from exc
|
|
except (json.JSONDecodeError, UnicodeDecodeError) as exc:
|
|
raise RuntimeError(f"Server at {target_url} returned an unparseable response: {exc}") from exc
|
|
|
|
|
|
def normalize_section_alias(value: str) -> str:
|
|
return re.sub(r"_+", "_", re.sub(r"[^a-z0-9]+", "_", value.lower())).strip("_")
|
|
|
|
|
|
def tool_id_candidates(tool_id: str) -> list[str]:
|
|
candidates = [tool_id]
|
|
if tool_id.startswith("toolshed.") and tool_id.count("/") >= 5:
|
|
short_id = tool_id.rsplit("/", 1)[0]
|
|
if short_id not in candidates:
|
|
candidates.append(short_id)
|
|
return candidates
|
|
|
|
|
|
def extract_sections_and_tools(data: list[dict[str, Any]]) -> tuple[dict[str, list[str]], dict[str, str]]:
|
|
"""Extract ToolSections and their tools from API response."""
|
|
sections: dict[str, list[str]] = {}
|
|
section_ids: dict[str, str] = {}
|
|
|
|
for item in data:
|
|
if item.get("model_class") == "ToolSection":
|
|
section_id = item.get("id")
|
|
section_name = item.get("name")
|
|
if section_id and section_name:
|
|
section_ids[section_id] = section_name
|
|
print(f"Found section: {section_name} (id: {section_id})")
|
|
|
|
# Extract tools in this section
|
|
tools = []
|
|
for elem in item.get("elems", []):
|
|
model_class = elem.get("model_class", "")
|
|
if isinstance(model_class, str) and model_class.endswith("Tool"):
|
|
tool_id = elem.get("id")
|
|
if tool_id:
|
|
for candidate in tool_id_candidates(tool_id):
|
|
if candidate not in tools:
|
|
tools.append(candidate)
|
|
|
|
if tools:
|
|
sections.setdefault(section_name, []).extend(tools)
|
|
print(f" - Contains {len(tools)} tools")
|
|
|
|
return sections, section_ids
|
|
|
|
|
|
def load_existing_mappings() -> dict[str, Any]:
|
|
"""Load existing tool_tag_mappings.yml"""
|
|
if OUTPUT_FILE.exists():
|
|
with open(OUTPUT_FILE, encoding="utf-8") as f:
|
|
return yaml.safe_load(f)
|
|
return {"tool_tags": {}}
|
|
|
|
|
|
def normalize_tool_tags(tool_tags: dict[str, list[str]]) -> dict[str, list[str]]:
|
|
normalized: dict[str, list[str]] = {}
|
|
for tool_id, tags in tool_tags.items():
|
|
unique_tags: list[str] = []
|
|
for tag in tags:
|
|
if tag not in unique_tags:
|
|
unique_tags.append(tag)
|
|
normalized[tool_id] = unique_tags
|
|
return normalized
|
|
|
|
|
|
def validate_mappings_data(data: dict[str, Any]) -> dict[str, dict[str, list[str]]]:
|
|
tool_tags = data.get("tool_tags", {})
|
|
if not isinstance(tool_tags, dict):
|
|
raise ValueError("tool_tag_mappings.yml must contain a top-level 'tool_tags' mapping.")
|
|
for tool_id, tags in tool_tags.items():
|
|
if not isinstance(tool_id, str):
|
|
raise ValueError(f"Tool id {tool_id!r} must be a string.")
|
|
if not isinstance(tags, list) or not all(isinstance(tag, str) for tag in tags):
|
|
raise ValueError(f"Tags for tool {tool_id!r} must be a list of strings.")
|
|
return {"tool_tags": normalize_tool_tags(tool_tags)}
|
|
|
|
|
|
def needs_quotes(value: str) -> bool:
|
|
return value in YAML_BOOLEAN_LIKE_VALUES or not SAFE_UNQUOTED_RE.fullmatch(value)
|
|
|
|
|
|
def prepare_for_dump(value: Any) -> Any:
|
|
if isinstance(value, dict):
|
|
return {prepare_for_dump(key): prepare_for_dump(item) for key, item in value.items()}
|
|
if isinstance(value, list):
|
|
return [prepare_for_dump(item) for item in value]
|
|
if isinstance(value, str) and needs_quotes(value):
|
|
return QuotedString(value)
|
|
return value
|
|
|
|
|
|
def update_mappings(
|
|
existing_data: dict[str, Any],
|
|
sections: dict[str, list[str]],
|
|
section_ids: dict[str, str],
|
|
) -> dict[str, dict[str, list[str]]]:
|
|
"""Refresh section tags while preserving existing non-section curated tags."""
|
|
section_aliases = {normalize_section_alias(alias) for alias in set(section_ids) | set(section_ids.values())}
|
|
tool_tags: dict[str, list[str]] = {}
|
|
|
|
for tool_id, tags in existing_data.get("tool_tags", {}).items():
|
|
custom_tags = [tag for tag in tags if normalize_section_alias(tag) not in section_aliases]
|
|
if custom_tags:
|
|
tool_tags[tool_id] = custom_tags
|
|
|
|
# Track statistics
|
|
new_mappings = 0
|
|
updated_mappings = 0
|
|
|
|
for section_name, tools in sections.items():
|
|
tag = section_name
|
|
|
|
for tool_id in tools:
|
|
if tool_id in tool_tags:
|
|
# Check if tag already exists
|
|
if tag not in tool_tags[tool_id]:
|
|
tool_tags[tool_id].append(tag)
|
|
updated_mappings += 1
|
|
print(f" Updated {tool_id} with tag: {tag}")
|
|
else:
|
|
tool_tags[tool_id] = [tag]
|
|
new_mappings += 1
|
|
print(f" Added {tool_id} with tag: {tag}")
|
|
|
|
print("\nSummary:")
|
|
print(f" - New tool mappings: {new_mappings}")
|
|
print(f" - Updated tool mappings: {updated_mappings}")
|
|
print(f" - Total tools with tags: {len(tool_tags)}")
|
|
|
|
return {"tool_tags": normalize_tool_tags(tool_tags)}
|
|
|
|
|
|
def save_mappings(data: dict[str, Any]) -> None:
|
|
"""Save updated tool_tag_mappings.yml"""
|
|
validated_data = validate_mappings_data(data)
|
|
serialized = yaml.dump(
|
|
prepare_for_dump(validated_data),
|
|
Dumper=QuotingDumper,
|
|
default_flow_style=False,
|
|
sort_keys=False,
|
|
allow_unicode=True,
|
|
width=1000,
|
|
)
|
|
|
|
# Validate that the serialized YAML round-trips cleanly.
|
|
round_trip = yaml.safe_load(serialized)
|
|
if round_trip != validated_data:
|
|
raise ValueError("Serialized YAML did not round-trip correctly.")
|
|
|
|
print(f"\nSaving to {OUTPUT_FILE}...")
|
|
with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
|
|
f.write(serialized)
|
|
print("Done!")
|
|
|
|
|
|
def _build_arg_parser() -> argparse.ArgumentParser:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument(
|
|
"--api-url",
|
|
default=DEFAULT_API_URL,
|
|
help=f"Galaxy /api/tools endpoint to scrape (default: {DEFAULT_API_URL}).",
|
|
)
|
|
parser.add_argument(
|
|
"--output",
|
|
type=Path,
|
|
default=DEFAULT_OUTPUT_FILE,
|
|
help=f"Path of the curated tool_tag_mappings.yml to write (default: {DEFAULT_OUTPUT_FILE}).",
|
|
)
|
|
parser.add_argument(
|
|
"--timeout",
|
|
type=float,
|
|
default=DEFAULT_TIMEOUT_SECONDS,
|
|
help=f"HTTP timeout in seconds (default: {DEFAULT_TIMEOUT_SECONDS}).",
|
|
)
|
|
return parser
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = _build_arg_parser().parse_args(argv)
|
|
|
|
global API_URL, OUTPUT_FILE
|
|
API_URL = args.api_url
|
|
OUTPUT_FILE = args.output
|
|
|
|
try:
|
|
data = fetch_tools(api_url=args.api_url, timeout=args.timeout)
|
|
except RuntimeError as exc:
|
|
print(f"Error: {exc}", file=sys.stderr)
|
|
return 2
|
|
|
|
sections, section_ids = extract_sections_and_tools(data)
|
|
if not sections:
|
|
print("No sections found!", file=sys.stderr)
|
|
return 1
|
|
|
|
existing_data = load_existing_mappings()
|
|
updated_data = update_mappings(existing_data, sections, section_ids)
|
|
save_mappings(updated_data)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|