Files
galaxy/scripts/extract_tool_sections_from_api.py
T
mvdbeek df4b4c8a12 Address gx-arch-review feedback on PR #22212
- Extract FavoritesManager (lib/galaxy/managers/favorites.py): owns
  normalize / resolve / persist / order logic; controllers in users.py
  are now 2 lines each, and trans.sa_session.commit moves out of the
  controller layer. Drops the redundant re-normalize on every write
  and replaces FavoriteObjectType.X.value == .value defensive dispatch
  with enum-identity comparison. Renames the set_favorite_order
  endpoint summary away from the ambiguous "top-level" wording.
- Move curated_tool_tags_by_id off the toolbox into ToolsService;
  the controller now routes through the service instead of poking
  trans.app.toolbox directly. Raw aggregates (curated_tool_tags,
  tool_edam_operations, tool_edam_topics) remain on the toolbox since
  they're derived from per-tool data.
- Rename /api/tools/tags -> /api/tags/tool_tags (operation_id
  tags__tool_tags) to disambiguate from /api/tools/{id}; update the
  client store, tests, and the hand-edited OpenAPI schema entry.
  (Run `make update-client-api-schema` before pushing to regenerate
  the schema in its canonical alphabetical position.)
- Promote inline imports to module-level (ontology_data.py logging,
  app/__init__.py configure_tool_tag_mapping); replace
  getattr(self.config, "tool_tag_mappings_file", None) with direct
  attribute access now that the option is registered in the schema.
- Convert ontology_data's import-time bundled-mapping reads to
  lru_cache so the YAML / TSV files load lazily; fix the
  Tool._setup_id comment to point at the actual lowercasing site
  (Tool.parse around `self_ids = [self.id.lower()]`).
- doc/source/admin/tool_panel.rst: 26.0 -> 26.1.
- scripts/extract_tool_sections_from_api.py: annotate the public
  helpers the test suite imports.
- test_to_dict_cache_drops_when_tool_removed now asserts observable
  to_dict() output before/after remove, not private cache membership.
- Split test_tool_discovery_landing into focused tests; extract a
  helper that opens the discovery view and types the standard filter.
- Move the Playwright drag helper from test_tool_panel_search.py into
  TestWithSeleniumMixin.playwright_drag_item_above so future tests
  reuse it.
- Client tests: drop the full-string Whoosh equality in
  utilities.test.ts; hard-code expected Whoosh strings in
  ToolsList.test.ts (drop createWhooshQuery as both sides of the
  assertion) and remove the redundant injectTestRouter mount option
  (ToolsList.vue only uses the useRouter composable); drop the
  FontAwesome data-icon/iconName assertions in ToolPanel.test.ts in
  favor of icon-existence checks; replace three near-identical
  add/remove favorite tests in ToolsListCard.test.ts with an it.each;
  export quoteToolTagValue from Panels/utilities so
  filterConversion.test.js imports the real helper instead of
  reconstructing it inline; delete the spy-only "loads the curated
  tag mapping" test in ToolBoxSearch.test.ts (the gate is covered by
  ToolPanel.test.ts's negative case and the favorite-tag rendering
  tests cover the positive case end-to-end).
2026-05-19 11:43:34 +02:00

303 lines
11 KiB
Python

#!/usr/bin/env python3
"""
Refresh section-derived tool tags from a Galaxy server's `/api/tools` payload.
Defaults to the EU Galaxy instance and writes the curated tag map to
``config/tool_tag_mappings.yml`` next to this repo; both can be overridden
via CLI flags so the script works from any cwd.
"""
import argparse
import json
import re
import sys
import urllib.error
import urllib.request
from pathlib import Path
from typing import (
Any,
Optional,
)
import yaml
from yaml.events import (
AliasEvent,
CollectionStartEvent,
NodeEvent,
ScalarEvent,
)
DEFAULT_API_URL = "https://usegalaxy.eu/api/tools/"
DEFAULT_OUTPUT_FILE = Path(__file__).resolve().parent.parent / "config" / "tool_tag_mappings.yml"
DEFAULT_TIMEOUT_SECONDS = 60
# Module-level globals overwritten from CLI args in ``main`` so the helper
# functions (which the test suite imports) keep their existing signatures.
API_URL = DEFAULT_API_URL
OUTPUT_FILE = DEFAULT_OUTPUT_FILE
SAFE_UNQUOTED_RE = re.compile(r"^[A-Za-z0-9_.-]+$")
YAML_BOOLEAN_LIKE_VALUES = {"", "~", "null", "Null", "NULL", "true", "True", "TRUE", "false", "False", "FALSE"}
class QuotedString(str):
"""Marker type for YAML strings that must be emitted with quotes."""
class QuotingDumper(yaml.SafeDumper):
"""YAML dumper that keeps scalar quoting explicit and predictable.
PyYAML's default dumper changes quoting heuristically when scalar lengths
cross internal thresholds, which churns the output for diff review every
time we regenerate the file. A round-trip dumper from ``ruamel.yaml`` would
avoid this, but ruamel.yaml isn't a Galaxy runtime dependency and this
script is admin-only, so we patch the simple-key heuristic instead.
"""
def check_simple_key(self):
length = 0
if isinstance(self.event, NodeEvent) and self.event.anchor is not None:
if self.prepared_anchor is None:
self.prepared_anchor = self.prepare_anchor(self.event.anchor)
length += len(self.prepared_anchor)
if isinstance(self.event, (ScalarEvent, CollectionStartEvent)) and self.event.tag is not None:
if self.prepared_tag is None:
self.prepared_tag = self.prepare_tag(self.event.tag)
length += len(self.prepared_tag)
if isinstance(self.event, ScalarEvent):
if self.analysis is None:
self.analysis = self.analyze_scalar(self.event.value)
length += len(self.analysis.scalar)
return length < 4096 and (
isinstance(self.event, AliasEvent)
or (isinstance(self.event, ScalarEvent) and not self.analysis.empty and not self.analysis.multiline)
or self.check_empty_sequence()
or self.check_empty_mapping()
)
def represent_quoted_string(dumper: yaml.Dumper, data: str) -> yaml.ScalarNode:
return dumper.represent_scalar("tag:yaml.org,2002:str", data, style='"')
QuotingDumper.add_representer(QuotedString, represent_quoted_string)
def fetch_tools(api_url: Optional[str] = None, timeout: float = DEFAULT_TIMEOUT_SECONDS) -> list[dict[str, Any]]:
"""Fetch tools from a Galaxy `/api/tools` endpoint.
Raises ``RuntimeError`` on transport / HTTP / decoding failures so the caller
can fail fast instead of silently overwriting the YAML with a no-op result.
"""
target_url = api_url or API_URL
print(f"Fetching tools from {target_url}...")
try:
with urllib.request.urlopen(target_url, timeout=timeout) as response:
return json.loads(response.read().decode("utf-8"))
except (urllib.error.URLError, TimeoutError) as exc:
raise RuntimeError(f"Could not reach {target_url}: {exc}") from exc
except (json.JSONDecodeError, UnicodeDecodeError) as exc:
raise RuntimeError(f"Server at {target_url} returned an unparseable response: {exc}") from exc
def normalize_section_alias(value: str) -> str:
return re.sub(r"_+", "_", re.sub(r"[^a-z0-9]+", "_", value.lower())).strip("_")
def tool_id_candidates(tool_id: str) -> list[str]:
candidates = [tool_id]
if tool_id.startswith("toolshed.") and tool_id.count("/") >= 5:
short_id = tool_id.rsplit("/", 1)[0]
if short_id not in candidates:
candidates.append(short_id)
return candidates
def extract_sections_and_tools(data: list[dict[str, Any]]) -> tuple[dict[str, list[str]], dict[str, str]]:
"""Extract ToolSections and their tools from API response."""
sections: dict[str, list[str]] = {}
section_ids: dict[str, str] = {}
for item in data:
if item.get("model_class") == "ToolSection":
section_id = item.get("id")
section_name = item.get("name")
if section_id and section_name:
section_ids[section_id] = section_name
print(f"Found section: {section_name} (id: {section_id})")
# Extract tools in this section
tools = []
for elem in item.get("elems", []):
model_class = elem.get("model_class", "")
if isinstance(model_class, str) and model_class.endswith("Tool"):
tool_id = elem.get("id")
if tool_id:
for candidate in tool_id_candidates(tool_id):
if candidate not in tools:
tools.append(candidate)
if tools:
sections.setdefault(section_name, []).extend(tools)
print(f" - Contains {len(tools)} tools")
return sections, section_ids
def load_existing_mappings() -> dict[str, Any]:
"""Load existing tool_tag_mappings.yml"""
if OUTPUT_FILE.exists():
with open(OUTPUT_FILE, encoding="utf-8") as f:
return yaml.safe_load(f)
return {"tool_tags": {}}
def normalize_tool_tags(tool_tags: dict[str, list[str]]) -> dict[str, list[str]]:
normalized: dict[str, list[str]] = {}
for tool_id, tags in tool_tags.items():
unique_tags: list[str] = []
for tag in tags:
if tag not in unique_tags:
unique_tags.append(tag)
normalized[tool_id] = unique_tags
return normalized
def validate_mappings_data(data: dict[str, Any]) -> dict[str, dict[str, list[str]]]:
tool_tags = data.get("tool_tags", {})
if not isinstance(tool_tags, dict):
raise ValueError("tool_tag_mappings.yml must contain a top-level 'tool_tags' mapping.")
for tool_id, tags in tool_tags.items():
if not isinstance(tool_id, str):
raise ValueError(f"Tool id {tool_id!r} must be a string.")
if not isinstance(tags, list) or not all(isinstance(tag, str) for tag in tags):
raise ValueError(f"Tags for tool {tool_id!r} must be a list of strings.")
return {"tool_tags": normalize_tool_tags(tool_tags)}
def needs_quotes(value: str) -> bool:
return value in YAML_BOOLEAN_LIKE_VALUES or not SAFE_UNQUOTED_RE.fullmatch(value)
def prepare_for_dump(value: Any) -> Any:
if isinstance(value, dict):
return {prepare_for_dump(key): prepare_for_dump(item) for key, item in value.items()}
if isinstance(value, list):
return [prepare_for_dump(item) for item in value]
if isinstance(value, str) and needs_quotes(value):
return QuotedString(value)
return value
def update_mappings(
existing_data: dict[str, Any],
sections: dict[str, list[str]],
section_ids: dict[str, str],
) -> dict[str, dict[str, list[str]]]:
"""Refresh section tags while preserving existing non-section curated tags."""
section_aliases = {normalize_section_alias(alias) for alias in set(section_ids) | set(section_ids.values())}
tool_tags: dict[str, list[str]] = {}
for tool_id, tags in existing_data.get("tool_tags", {}).items():
custom_tags = [tag for tag in tags if normalize_section_alias(tag) not in section_aliases]
if custom_tags:
tool_tags[tool_id] = custom_tags
# Track statistics
new_mappings = 0
updated_mappings = 0
for section_name, tools in sections.items():
tag = section_name
for tool_id in tools:
if tool_id in tool_tags:
# Check if tag already exists
if tag not in tool_tags[tool_id]:
tool_tags[tool_id].append(tag)
updated_mappings += 1
print(f" Updated {tool_id} with tag: {tag}")
else:
tool_tags[tool_id] = [tag]
new_mappings += 1
print(f" Added {tool_id} with tag: {tag}")
print("\nSummary:")
print(f" - New tool mappings: {new_mappings}")
print(f" - Updated tool mappings: {updated_mappings}")
print(f" - Total tools with tags: {len(tool_tags)}")
return {"tool_tags": normalize_tool_tags(tool_tags)}
def save_mappings(data: dict[str, Any]) -> None:
"""Save updated tool_tag_mappings.yml"""
validated_data = validate_mappings_data(data)
serialized = yaml.dump(
prepare_for_dump(validated_data),
Dumper=QuotingDumper,
default_flow_style=False,
sort_keys=False,
allow_unicode=True,
width=1000,
)
# Validate that the serialized YAML round-trips cleanly.
round_trip = yaml.safe_load(serialized)
if round_trip != validated_data:
raise ValueError("Serialized YAML did not round-trip correctly.")
print(f"\nSaving to {OUTPUT_FILE}...")
with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
f.write(serialized)
print("Done!")
def _build_arg_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--api-url",
default=DEFAULT_API_URL,
help=f"Galaxy /api/tools endpoint to scrape (default: {DEFAULT_API_URL}).",
)
parser.add_argument(
"--output",
type=Path,
default=DEFAULT_OUTPUT_FILE,
help=f"Path of the curated tool_tag_mappings.yml to write (default: {DEFAULT_OUTPUT_FILE}).",
)
parser.add_argument(
"--timeout",
type=float,
default=DEFAULT_TIMEOUT_SECONDS,
help=f"HTTP timeout in seconds (default: {DEFAULT_TIMEOUT_SECONDS}).",
)
return parser
def main(argv: Optional[list[str]] = None) -> int:
args = _build_arg_parser().parse_args(argv)
global API_URL, OUTPUT_FILE
API_URL = args.api_url
OUTPUT_FILE = args.output
try:
data = fetch_tools(api_url=args.api_url, timeout=args.timeout)
except RuntimeError as exc:
print(f"Error: {exc}", file=sys.stderr)
return 2
sections, section_ids = extract_sections_and_tools(data)
if not sections:
print("No sections found!", file=sys.stderr)
return 1
existing_data = load_existing_mappings()
updated_data = update_mappings(existing_data, sections, section_ids)
save_mappings(updated_data)
return 0
if __name__ == "__main__":
sys.exit(main())