From 1c0a47a9d2b314cb763904c23e6be9605a2ecdc3 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 12:49:26 -0400 Subject: [PATCH 01/13] Add gtn_database_refresh_interval so admins can tune (or disable, with 0) how often Galaxy re-pulls the GTN search database from depot. The periodic celery task that actually does the refresh follows in a separate commit. --- doc/source/admin/galaxy_options.rst | 88 ++++++++++++++++++++- lib/galaxy/config/sample/galaxy.yml.sample | 33 ++++++-- lib/galaxy/config/schemas/config_schema.yml | 13 +++ 3 files changed, 123 insertions(+), 11 deletions(-) diff --git a/doc/source/admin/galaxy_options.rst b/doc/source/admin/galaxy_options.rst index 6b285e7cf4e..e06967bd139 100644 --- a/doc/source/admin/galaxy_options.rst +++ b/doc/source/admin/galaxy_options.rst @@ -1305,6 +1305,24 @@ :Type: str +~~~~~~~~~~~~~~~~~~~~~~~~~~ +``tool_tag_mappings_file`` +~~~~~~~~~~~~~~~~~~~~~~~~~~ + +:Description: + Optional YAML file mapping tool ids to curated tag names. Tags + drive the `tag:` autocompletion, the favorite-tags grouping in the + My Tools panel, and the `tool_tags` Whoosh search field. If unset, + Galaxy ships a small example covering the tools used by its own + integration tests; admins who want production-grade tag coverage + can generate a snapshot for their instance with + `scripts/extract_tool_sections_from_api.py`. The file must be a + YAML document with a top-level `tool_tags:` mapping from tool id + to a list of tag names. +:Default: ``None`` +:Type: str + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ ``biotools_content_directory`` ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -5663,11 +5681,76 @@ false }, jupyterlite: { model: gpt-4o } } Set static_responses to a YAML file path to replace all LLM calls with deterministic responses for testing: inference_services: { static_responses: - test/integration/static_agents.yml } + test/integration/static_agents.yml } Per-agent or default-block + ``structured_output_override: true|false`` beats the model + capability table -- see ``agent_model_capabilities_file`` for the + table's location and contents. :Default: ``None`` :Type: any +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +``agent_model_capabilities_file`` +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +:Description: + YAML file with capability hints for agent inference models. Maps + fnmatch-style globs against model names to features such as + structured-output (tool-calling / JSON-mode) support. Galaxy ships + a sample populated with common model families; admins can drop a + file named ``agent_model_capabilities.yml`` in ``config_dir`` to + override the shipped table for private models. + ``inference_services`` ``structured_output_override`` overrides + this table for a specific agent or default block. + The value of this option will be resolved with respect to + . +:Default: ``agent_model_capabilities.yml`` +:Type: str + + +~~~~~~~~~~~~~~~~~~~~~ +``gtn_database_path`` +~~~~~~~~~~~~~~~~~~~~~ + +:Description: + Path to the SQLite FTS5 database used by the GTN training agent. + Resolves against ``data_dir`` so admins can place it in a + mutable-data directory. The file is downloaded automatically on + first use from ``gtn_database_url`` if it does not exist. + The value of this option will be resolved with respect to + . +:Default: ``gtn/gtn_search.db`` +:Type: str + + +~~~~~~~~~~~~~~~~~~~~ +``gtn_database_url`` +~~~~~~~~~~~~~~~~~~~~ + +:Description: + URL used to download the GTN search database when the local file + at ``gtn_database_path`` is missing. +:Default: ``https://depot.galaxyproject.org/chatgxy/gtn_search.db`` +:Type: str + + +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +``gtn_database_refresh_interval`` +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +:Description: + Time (in seconds) between celery-beat triggered refreshes of the + GTN search database from ``gtn_database_url``. The task + re-downloads the file and atomically replaces + ``gtn_database_path``; live handlers pick up the new copy on their + next query since GTNSearchDB opens a read-only connection per + call. Set to 0 to disable automatic refresh (admins can still + refresh on demand via ``python -m galaxy.agents.gtn --refresh``). + Requires celery. +:Default: ``3600`` +:Type: int + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ ``enable_tool_recommendations`` ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -6163,6 +6246,3 @@ for user defined tools. :Default: ``false`` :Type: bool - - - diff --git a/lib/galaxy/config/sample/galaxy.yml.sample b/lib/galaxy/config/sample/galaxy.yml.sample index d8cdb8b7d06..ced13d8d8ca 100644 --- a/lib/galaxy/config/sample/galaxy.yml.sample +++ b/lib/galaxy/config/sample/galaxy.yml.sample @@ -1,21 +1,21 @@ # Galaxy is configured by default to be usable in a single-user development # environment. To tune the application for a multi-user production # environment, see the documentation at: -# +# # https://docs.galaxyproject.org/en/master/admin/production.html -# +# # Throughout this sample configuration file, except where stated otherwise, # uncommented values override the default if left unset, whereas commented # values are set to the default value. Relative paths are relative to the root # Galaxy directory. -# +# # Examples of many of these options are explained in more detail in the Galaxy # Community Hub. -# +# # https://galaxyproject.org/admin/config -# +# # Config hackers are encouraged to check there before asking for help. -# +# # Configuration for Gravity process manager. # ``uwsgi:`` section will be ignored if Galaxy is started via Gravity commands (e.g ``./run.sh``, ``galaxy`` or ``galaxyctl``). gravity: @@ -1007,6 +1007,17 @@ galaxy: # this option will be resolved with respect to . #tool_search_index_dir: tool_search_index + # Optional YAML file mapping tool ids to curated tag names. Tags drive + # the `tag:` autocompletion, the favorite-tags grouping in the My + # Tools panel, and the `tool_tags` Whoosh search field. If unset, + # Galaxy ships a small example covering the tools used by its own + # integration tests; admins who want production-grade tag coverage can + # generate a snapshot for their instance with + # `scripts/extract_tool_sections_from_api.py`. The file must be a YAML + # document with a top-level `tool_tags:` mapping from tool id to a + # list of tag names. + #tool_tag_mappings_file: null + # Point Galaxy at a repository consisting of a copy of the bio.tools # database (e.g. https://github.com/bio-tools/content/) to resolve # bio.tools data for tool metadata. @@ -3096,6 +3107,15 @@ galaxy: # ``gtn_database_path`` is missing. #gtn_database_url: https://depot.galaxyproject.org/chatgxy/gtn_search.db + # Time (in seconds) between celery-beat triggered refreshes of the GTN + # search database from ``gtn_database_url``. The task re-downloads the + # file and atomically replaces ``gtn_database_path``; live handlers + # pick up the new copy on their next query since GTNSearchDB opens a + # read-only connection per call. Set to 0 to disable automatic refresh + # (admins can still refresh on demand via ``python -m + # galaxy.agents.gtn --refresh``). Requires celery. + #gtn_database_refresh_interval: 3600 + # Allow the display of tool recommendations in workflow editor and # after tool execution. If it is enabled and set to true, please # enable 'tool_recommendation_model_path' as well @@ -3330,4 +3350,3 @@ galaxy: # Enable beta tool formats (yaml, cwl, ...) which is a prerequisite # for user defined tools. #enable_beta_tool_formats: false - diff --git a/lib/galaxy/config/schemas/config_schema.yml b/lib/galaxy/config/schemas/config_schema.yml index c16bc6901b0..4ab687d8285 100644 --- a/lib/galaxy/config/schemas/config_schema.yml +++ b/lib/galaxy/config/schemas/config_schema.yml @@ -4243,6 +4243,19 @@ mapping: URL used to download the GTN search database when the local file at ``gtn_database_path`` is missing. + gtn_database_refresh_interval: + type: int + default: 3600 + required: false + desc: | + Time (in seconds) between celery-beat triggered refreshes of the GTN + search database from ``gtn_database_url``. The task re-downloads the + file and atomically replaces ``gtn_database_path``; live handlers + pick up the new copy on their next query since GTNSearchDB opens a + read-only connection per call. Set to 0 to disable automatic + refresh (admins can still refresh on demand via + ``python -m galaxy.agents.gtn --refresh``). Requires celery. + enable_tool_recommendations: type: bool default: false From 73bd73cdfd4f3184c77d6fdcacdd90ba4b9001c4 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 12:49:32 -0400 Subject: [PATCH 02/13] Refresh the GTN search database on a celery beat schedule. The task re-downloads from depot to a tmp path, validates, and atomically renames over the live copy -- handlers pick up the new file on the next query since GTNSearchDB opens read-only per call. Registration is gated on both the interval being non-zero and a configured gtn_database_path, so disabling the GTN agent doesn't leave a dangling periodic task. --- lib/galaxy/celery/__init__.py | 3 +++ lib/galaxy/celery/tasks.py | 28 ++++++++++++++++++++++++++++ test/unit/app/test_celery.py | 4 ++++ 3 files changed, 35 insertions(+) diff --git a/lib/galaxy/celery/__init__.py b/lib/galaxy/celery/__init__.py index b580f4340f7..122a44f16a9 100644 --- a/lib/galaxy/celery/__init__.py +++ b/lib/galaxy/celery/__init__.py @@ -283,6 +283,9 @@ def setup_periodic_tasks(config, celery_app): if config.vault_token_renewal_interval: schedule_task("renew_vault_token", config.vault_token_renewal_interval) + if getattr(config, "gtn_database_refresh_interval", 0) and getattr(config, "gtn_database_path", None): + schedule_task("refresh_gtn_database", config.gtn_database_refresh_interval) + if config.celery_user_concurrency_limit: # Run cleanup every 5 minutes (300 seconds) schedule_task("cleanup_stale_concurrency_slots", 300) diff --git a/lib/galaxy/celery/tasks.py b/lib/galaxy/celery/tasks.py index 041b8201f49..41f9395f6e7 100644 --- a/lib/galaxy/celery/tasks.py +++ b/lib/galaxy/celery/tasks.py @@ -26,6 +26,7 @@ from sqlalchemy import ( ) from galaxy import model +from galaxy.agents.gtn import GTNSearchDB from galaxy.celery import ( celery_app, galaxy_task, @@ -829,6 +830,33 @@ def renew_vault_token(vault: Vault): renew_vault_token_if_needed(vault) +@galaxy_task(action="refreshing GTN training database") +def refresh_gtn_database(config: GalaxyAppConfiguration): + """Re-download the GTN search database from depot, replacing the local copy atomically. + + Handlers open the database read-only per query, so an atomic rename here + is picked up by the next ChatGXY request without a restart. Failures are + logged and swallowed so a depot outage doesn't kill the periodic queue. + """ + db_path = getattr(config, "gtn_database_path", None) + if not db_path: + log.debug("refresh_gtn_database: gtn_database_path is unset, skipping") + return + download_url = getattr(config, "gtn_database_url", None) + try: + metadata = GTNSearchDB.refresh_database(db_path, download_url) + except FileNotFoundError as e: + log.warning("refresh_gtn_database: download failed, keeping existing copy: %s", e) + return + log.info( + "refresh_gtn_database: refreshed %s (version=%s, tutorials=%s, faqs=%s)", + db_path, + metadata["version"], + metadata["tutorial_count"], + metadata["faq_count"], + ) + + @galaxy_task(action="execute workflow completion hook") def execute_workflow_completion_hook( invocation_id: int, diff --git a/test/unit/app/test_celery.py b/test/unit/app/test_celery.py index db16b23b4b8..490d3ac603d 100644 --- a/test/unit/app/test_celery.py +++ b/test/unit/app/test_celery.py @@ -26,6 +26,10 @@ def test_default_configuration(): "task": "galaxy.cleanup_short_term_storage", "schedule": galaxy_conf.short_term_storage_cleanup_interval, } + assert conf.beat_schedule["refresh-gtn-database"] == { + "task": "galaxy.refresh_gtn_database", + "schedule": galaxy_conf.gtn_database_refresh_interval, + } def test_galaxycelery_trim_module_name(): From 8875431e1a6a04b7b3f490f787062c17c08f4f04 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 13:48:08 -0400 Subject: [PATCH 03/13] Gate the GTN refresh schedule on inference_services being configured. Without it, the default config would have every Galaxy install pulling from depot on the schedule even though most don't use ChatGXY. Same predicate the agent registry already uses to decide whether the GTN agent is registered at all. --- lib/galaxy/celery/__init__.py | 10 +++++++++- test/unit/app/test_celery.py | 15 +++++++++++++-- 2 files changed, 22 insertions(+), 3 deletions(-) diff --git a/lib/galaxy/celery/__init__.py b/lib/galaxy/celery/__init__.py index 122a44f16a9..956d4885eed 100644 --- a/lib/galaxy/celery/__init__.py +++ b/lib/galaxy/celery/__init__.py @@ -283,7 +283,15 @@ def setup_periodic_tasks(config, celery_app): if config.vault_token_renewal_interval: schedule_task("renew_vault_token", config.vault_token_renewal_interval) - if getattr(config, "gtn_database_refresh_interval", 0) and getattr(config, "gtn_database_path", None): + # Only schedule if ChatGXY infrastructure is configured -- the GTN database + # serves the gtn_training agent, which only functions when inference_services + # is set up. Without this gate every Galaxy install would pull from depot + # on the default interval, even ones that don't use the agent. + if ( + getattr(config, "inference_services", None) + and config.gtn_database_refresh_interval + and config.gtn_database_path + ): schedule_task("refresh_gtn_database", config.gtn_database_refresh_interval) if config.celery_user_concurrency_limit: diff --git a/test/unit/app/test_celery.py b/test/unit/app/test_celery.py index 490d3ac603d..48dbc02ea6d 100644 --- a/test/unit/app/test_celery.py +++ b/test/unit/app/test_celery.py @@ -2,6 +2,7 @@ from galaxy.celery import ( celery_app, DEFAULT_TASK_QUEUE, GalaxyCelery, + setup_periodic_tasks, TASKS_MODULES, ) from galaxy.config import GalaxyAppConfiguration @@ -26,9 +27,19 @@ def test_default_configuration(): "task": "galaxy.cleanup_short_term_storage", "schedule": galaxy_conf.short_term_storage_cleanup_interval, } - assert conf.beat_schedule["refresh-gtn-database"] == { + # GTN refresh is gated on inference_services being configured; default + # config doesn't set it, so the schedule isn't registered here. + assert "refresh-gtn-database" not in conf.beat_schedule + + +def test_gtn_refresh_schedules_when_inference_configured(): + config = GalaxyAppConfiguration(override_tempdir=False) + config.inference_services = {"default": {"model": "test"}} + app = GalaxyCelery("test-gtn-schedule") + setup_periodic_tasks(config, app) + assert app.conf.beat_schedule["refresh-gtn-database"] == { "task": "galaxy.refresh_gtn_database", - "schedule": galaxy_conf.gtn_database_refresh_interval, + "schedule": config.gtn_database_refresh_interval, } From 464a9a0b83d84c276df54c2e1e254980b8dae78c Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 13:49:24 -0400 Subject: [PATCH 04/13] HEAD depot first and only re-download when its Last-Modified is newer than the local file. After each successful download the local mtime is stamped to depot's Last-Modified so future comparisons are exact and survive local clock skew. Raises the default refresh interval to once a day -- the dataset updates weekly or so, and the HEAD check keeps steady-state cost to a few hundred bytes per tick regardless. --- doc/source/admin/galaxy_options.rst | 17 ++-- lib/galaxy/agents/gtn/search.py | 70 ++++++++++++++++ lib/galaxy/celery/tasks.py | 17 ++-- lib/galaxy/config/sample/galaxy.yml.sample | 20 +++-- lib/galaxy/config/schemas/config_schema.yml | 20 +++-- test/unit/app/test_gtn_search.py | 90 +++++++++++++++++++++ 6 files changed, 205 insertions(+), 29 deletions(-) diff --git a/doc/source/admin/galaxy_options.rst b/doc/source/admin/galaxy_options.rst index e06967bd139..4608e85d32b 100644 --- a/doc/source/admin/galaxy_options.rst +++ b/doc/source/admin/galaxy_options.rst @@ -5739,15 +5739,18 @@ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ :Description: - Time (in seconds) between celery-beat triggered refreshes of the - GTN search database from ``gtn_database_url``. The task - re-downloads the file and atomically replaces + Time (in seconds) between celery-beat triggered freshness checks + of the GTN search database from ``gtn_database_url``. The task + HEADs depot and only re-downloads when its ``Last-Modified`` is + newer than the local file -- steady-state cost is a few hundred + bytes per tick. When a download does happen it atomically replaces ``gtn_database_path``; live handlers pick up the new copy on their next query since GTNSearchDB opens a read-only connection per - call. Set to 0 to disable automatic refresh (admins can still - refresh on demand via ``python -m galaxy.agents.gtn --refresh``). - Requires celery. -:Default: ``3600`` + call. Only registered when ``inference_services`` is configured + (i.e. ChatGXY is in use). Set to 0 to disable automatic refresh -- + admins can still refresh on demand via ``python -m + galaxy.agents.gtn --refresh``. Requires celery. +:Default: ``86400`` :Type: int diff --git a/lib/galaxy/agents/gtn/search.py b/lib/galaxy/agents/gtn/search.py index 04da9448130..d4ed28d2386 100644 --- a/lib/galaxy/agents/gtn/search.py +++ b/lib/galaxy/agents/gtn/search.py @@ -6,11 +6,18 @@ using SQLite FTS5 full-text search with BM25 ranking. """ import logging +import os import re import shutil import sqlite3 import urllib.request from dataclasses import dataclass +from datetime import ( + datetime, + timedelta, + timezone, +) +from email.utils import parsedate_to_datetime from pathlib import Path from typing import ( Any, @@ -24,10 +31,28 @@ GTN_FAQ_BASE_URL = "https://training.galaxyproject.org/training-material/faqs" # an agent init forever. Total wall-clock can still exceed this if the # remote keeps sending small chunks, which is the trade for stdlib-only. GTN_DOWNLOAD_TIMEOUT_SECONDS = 60 +# HEAD is a small request; an unreachable depot during a freshness check +# shouldn't hold the periodic queue for the full download budget. +GTN_FRESHNESS_TIMEOUT_SECONDS = 10 + +_ONE_SECOND = timedelta(seconds=1) log = logging.getLogger(__name__) +def _parse_last_modified(header: Optional[str]) -> Optional[datetime]: + """Parse an HTTP Last-Modified header into an aware UTC datetime, or None.""" + if not header: + return None + try: + parsed = parsedate_to_datetime(header) + except (TypeError, ValueError): + return None + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed + + def _escape_like(value: str) -> str: """Escape SQLite LIKE metacharacters so tool names match literally.""" return value.replace("\\", "\\\\").replace("%", r"\%").replace("_", r"\_") @@ -194,6 +219,43 @@ class GTNSearchDB: """Download, validate, and atomically replace a GTN database without opening the old copy.""" return cls._download_database_to_path(Path(db_path), download_url or GTN_DATABASE_URL) + @classmethod + def refresh_database_if_stale( + cls, db_path: str | Path, download_url: Optional[str] = None + ) -> Optional[dict[str, Any]]: + """HEAD the URL and re-download only if its Last-Modified is newer than the local file's mtime. + + Returns the new metadata dict when a refresh happened, ``None`` when the + local copy was current. If the local file is missing, or the HEAD fails, + falls through to a full download -- safer than skipping silently when we + can't tell whether the local copy is current. + """ + target = Path(db_path) + url = download_url or GTN_DATABASE_URL + + if target.exists(): + remote_mtime = cls._remote_last_modified(url) + if remote_mtime is not None: + local_mtime = datetime.fromtimestamp(target.stat().st_mtime, tz=timezone.utc) + # depot's Last-Modified has second resolution; the 1-second slack + # absorbs rounding so a successful refresh isn't immediately re-triggered. + if remote_mtime <= local_mtime + _ONE_SECOND: + return None + + return cls._download_database_to_path(target, url) + + @staticmethod + def _remote_last_modified(url: str) -> Optional[datetime]: + """HEAD ``url`` and return its parsed Last-Modified, or None on failure.""" + try: + req = urllib.request.Request(url, method="HEAD") + with urllib.request.urlopen(req, timeout=GTN_FRESHNESS_TIMEOUT_SECONDS) as resp: + header = resp.headers.get("Last-Modified") + except (OSError, ValueError) as e: + log.debug(f"GTN freshness HEAD failed for {url}: {e}") + return None + return _parse_last_modified(header) + @classmethod def _download_database_to_path(cls, db_path: Path, download_url: str) -> dict[str, Any]: db_path.parent.mkdir(parents=True, exist_ok=True) @@ -202,9 +264,17 @@ class GTNSearchDB: try: log.info(f"Downloading GTN database from {download_url} ...") with urllib.request.urlopen(download_url, timeout=GTN_DOWNLOAD_TIMEOUT_SECONDS) as response: + last_modified_header = response.headers.get("Last-Modified") with open(tmp_path, "wb") as out: shutil.copyfileobj(response, out) metadata = cls._validate_database_file(tmp_path) + # Stamp the file with depot's Last-Modified so the next stale check + # compares against the upstream mtime rather than "right now", which + # would also drift with any local clock skew. + remote_dt = _parse_last_modified(last_modified_header) + if remote_dt is not None: + remote_ts = remote_dt.timestamp() + os.utime(tmp_path, (remote_ts, remote_ts)) tmp_path.replace(db_path) return metadata except (OSError, sqlite3.Error) as e: diff --git a/lib/galaxy/celery/tasks.py b/lib/galaxy/celery/tasks.py index 41f9395f6e7..44ec4f7db72 100644 --- a/lib/galaxy/celery/tasks.py +++ b/lib/galaxy/celery/tasks.py @@ -832,22 +832,27 @@ def renew_vault_token(vault: Vault): @galaxy_task(action="refreshing GTN training database") def refresh_gtn_database(config: GalaxyAppConfiguration): - """Re-download the GTN search database from depot, replacing the local copy atomically. + """HEAD depot for the GTN search database and re-download only when newer. Handlers open the database read-only per query, so an atomic rename here - is picked up by the next ChatGXY request without a restart. Failures are - logged and swallowed so a depot outage doesn't kill the periodic queue. + is picked up by the next ChatGXY request without a restart. The HEAD-first + pattern keeps the steady-state cost to a few hundred bytes per tick -- + only when depot has actually been updated do we pull the full ~17MB. + Failures are logged and swallowed so a depot outage doesn't kill the + periodic queue. """ - db_path = getattr(config, "gtn_database_path", None) + db_path = config.gtn_database_path if not db_path: log.debug("refresh_gtn_database: gtn_database_path is unset, skipping") return - download_url = getattr(config, "gtn_database_url", None) try: - metadata = GTNSearchDB.refresh_database(db_path, download_url) + metadata = GTNSearchDB.refresh_database_if_stale(db_path, config.gtn_database_url) except FileNotFoundError as e: log.warning("refresh_gtn_database: download failed, keeping existing copy: %s", e) return + if metadata is None: + log.debug("refresh_gtn_database: %s is current, no download needed", db_path) + return log.info( "refresh_gtn_database: refreshed %s (version=%s, tutorials=%s, faqs=%s)", db_path, diff --git a/lib/galaxy/config/sample/galaxy.yml.sample b/lib/galaxy/config/sample/galaxy.yml.sample index ced13d8d8ca..dec50bd87a9 100644 --- a/lib/galaxy/config/sample/galaxy.yml.sample +++ b/lib/galaxy/config/sample/galaxy.yml.sample @@ -3107,14 +3107,18 @@ galaxy: # ``gtn_database_path`` is missing. #gtn_database_url: https://depot.galaxyproject.org/chatgxy/gtn_search.db - # Time (in seconds) between celery-beat triggered refreshes of the GTN - # search database from ``gtn_database_url``. The task re-downloads the - # file and atomically replaces ``gtn_database_path``; live handlers - # pick up the new copy on their next query since GTNSearchDB opens a - # read-only connection per call. Set to 0 to disable automatic refresh - # (admins can still refresh on demand via ``python -m - # galaxy.agents.gtn --refresh``). Requires celery. - #gtn_database_refresh_interval: 3600 + # Time (in seconds) between celery-beat triggered freshness checks of + # the GTN search database from ``gtn_database_url``. The task HEADs + # depot and only re-downloads when its ``Last-Modified`` is newer than + # the local file -- steady-state cost is a few hundred bytes per tick. + # When a download does happen it atomically replaces + # ``gtn_database_path``; live handlers pick up the new copy on their + # next query since GTNSearchDB opens a read-only connection per call. + # Only registered when ``inference_services`` is configured (i.e. + # ChatGXY is in use). Set to 0 to disable automatic refresh -- admins + # can still refresh on demand via ``python -m galaxy.agents.gtn + # --refresh``. Requires celery. + #gtn_database_refresh_interval: 86400 # Allow the display of tool recommendations in workflow editor and # after tool execution. If it is enabled and set to true, please diff --git a/lib/galaxy/config/schemas/config_schema.yml b/lib/galaxy/config/schemas/config_schema.yml index 4ab687d8285..4b7e493fd35 100644 --- a/lib/galaxy/config/schemas/config_schema.yml +++ b/lib/galaxy/config/schemas/config_schema.yml @@ -4245,16 +4245,20 @@ mapping: gtn_database_refresh_interval: type: int - default: 3600 + default: 86400 required: false desc: | - Time (in seconds) between celery-beat triggered refreshes of the GTN - search database from ``gtn_database_url``. The task re-downloads the - file and atomically replaces ``gtn_database_path``; live handlers - pick up the new copy on their next query since GTNSearchDB opens a - read-only connection per call. Set to 0 to disable automatic - refresh (admins can still refresh on demand via - ``python -m galaxy.agents.gtn --refresh``). Requires celery. + Time (in seconds) between celery-beat triggered freshness checks of + the GTN search database from ``gtn_database_url``. The task HEADs + depot and only re-downloads when its ``Last-Modified`` is newer than + the local file -- steady-state cost is a few hundred bytes per tick. + When a download does happen it atomically replaces + ``gtn_database_path``; live handlers pick up the new copy on their + next query since GTNSearchDB opens a read-only connection per call. + Only registered when ``inference_services`` is configured (i.e. + ChatGXY is in use). Set to 0 to disable automatic refresh -- admins + can still refresh on demand via + ``python -m galaxy.agents.gtn --refresh``. Requires celery. enable_tool_recommendations: type: bool diff --git a/test/unit/app/test_gtn_search.py b/test/unit/app/test_gtn_search.py index d33c458627a..43dd72f1d9b 100644 --- a/test/unit/app/test_gtn_search.py +++ b/test/unit/app/test_gtn_search.py @@ -5,7 +5,13 @@ result types against tiny in-memory fixture databases built via GTNDatabaseBuilder. """ +import os import sqlite3 +from datetime import ( + datetime, + timedelta, + timezone, +) from pathlib import Path import pytest @@ -217,3 +223,87 @@ def test_refresh_database_rejects_invalid_payload(tmp_path: Path): # Target was not replaced with the bogus payload. assert target.read_text(errors="replace") != "not a sqlite database" + + +def _build_fixture_db(target: Path, marker_title: str) -> None: + """Build a tiny valid GTN DB whose only tutorial title is ``marker_title``.""" + builder = GTNDatabaseBuilder(gtn_path=target.parent, output_path=target) + builder.tutorials = [ + Tutorial( + topic="freshness", + tutorial="marker", + title=marker_title, + description="", + url="https://example.invalid", + content="content", + content_hash="hash", + ) + ] + builder.create_database() + builder.insert_tutorials() + builder.add_metadata() + + +def _tutorial_titles(db_path: Path) -> list[str]: + with sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) as conn: + return [row[0] for row in conn.execute("SELECT title FROM tutorials")] + + +def test_refresh_database_if_stale_downloads_when_remote_is_newer(tmp_path: Path): + remote_path = tmp_path / "remote.db" + local_path = tmp_path / "local.db" + _build_fixture_db(remote_path, "REMOTE") + _build_fixture_db(local_path, "LOCAL") + + # Push local mtime well into the past so remote is unambiguously newer. + past = (datetime.now(timezone.utc) - timedelta(days=30)).timestamp() + os.utime(local_path, (past, past)) + + metadata = GTNSearchDB.refresh_database_if_stale(local_path, remote_path.as_uri()) + + assert metadata is not None + assert _tutorial_titles(local_path) == ["REMOTE"] + + +def test_refresh_database_if_stale_skips_when_local_is_current(tmp_path: Path): + remote_path = tmp_path / "remote.db" + local_path = tmp_path / "local.db" + _build_fixture_db(remote_path, "REMOTE") + _build_fixture_db(local_path, "LOCAL") + + # Push remote into the past so local always looks current. + past = (datetime.now(timezone.utc) - timedelta(days=30)).timestamp() + os.utime(remote_path, (past, past)) + + local_mtime_before = local_path.stat().st_mtime + metadata = GTNSearchDB.refresh_database_if_stale(local_path, remote_path.as_uri()) + + assert metadata is None + assert _tutorial_titles(local_path) == ["LOCAL"] + assert local_path.stat().st_mtime == local_mtime_before + + +def test_refresh_database_if_stale_downloads_when_local_is_missing(tmp_path: Path): + remote_path = tmp_path / "remote.db" + _build_fixture_db(remote_path, "REMOTE") + local_path = tmp_path / "cold-start.db" + + metadata = GTNSearchDB.refresh_database_if_stale(local_path, remote_path.as_uri()) + + assert metadata is not None + assert local_path.exists() + assert _tutorial_titles(local_path) == ["REMOTE"] + + +def test_download_stamps_local_mtime_to_remote_last_modified(tmp_path: Path): + remote_path = tmp_path / "remote.db" + _build_fixture_db(remote_path, "REMOTE") + # Pin the remote file's mtime to a known past second so we can assert + # the local copy ends up with the same second after download. + pinned = (datetime.now(timezone.utc) - timedelta(days=7)).replace(microsecond=0).timestamp() + os.utime(remote_path, (pinned, pinned)) + + local_path = tmp_path / "local.db" + GTNSearchDB.refresh_database(local_path, remote_path.as_uri()) + + assert int(local_path.stat().st_mtime) == int(pinned) From 6cbeadef9e49dda04d30de7f6c84fd58811769ce Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 16:28:06 -0400 Subject: [PATCH 05/13] Broaden the freshness HEAD catch to include http.client.HTTPException. Malformed responses from depot (RemoteDisconnected, BadStatusLine) aren't subclasses of OSError so they'd escape past the (OSError, ValueError) catch, propagate through refresh_database_if_stale to the periodic task, and record a failed run instead of falling through to a full download. --- lib/galaxy/agents/gtn/search.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/lib/galaxy/agents/gtn/search.py b/lib/galaxy/agents/gtn/search.py index d4ed28d2386..77aad0a543d 100644 --- a/lib/galaxy/agents/gtn/search.py +++ b/lib/galaxy/agents/gtn/search.py @@ -5,6 +5,7 @@ Provides search over Galaxy Training Network tutorials and FAQs using SQLite FTS5 full-text search with BM25 ranking. """ +import http.client import logging import os import re @@ -251,7 +252,11 @@ class GTNSearchDB: req = urllib.request.Request(url, method="HEAD") with urllib.request.urlopen(req, timeout=GTN_FRESHNESS_TIMEOUT_SECONDS) as resp: header = resp.headers.get("Last-Modified") - except (OSError, ValueError) as e: + except (OSError, ValueError, http.client.HTTPException) as e: + # http.client.HTTPException covers malformed responses + # (RemoteDisconnected, BadStatusLine) that urllib doesn't wrap + # into URLError -- without it the periodic queue would record a + # failed run instead of falling through to a full download. log.debug(f"GTN freshness HEAD failed for {url}: {e}") return None return _parse_last_modified(header) From fdfa892459be7a189ec0b7b55a5136e9a13dde83 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 17:17:36 -0400 Subject: [PATCH 06/13] Add iwc_manifest_refresh_interval so admins can tune (or disable, with 0) how often Galaxy pre-warms the IWC workflow manifest cache. Default of 3600 matches the in-process TTL so the cache stays continuously warm rather than expiring between user-driven hits. The periodic celery task follows in a separate commit. --- doc/source/admin/galaxy_options.rst | 18 ++++++++++++++++++ lib/galaxy/config/sample/galaxy.yml.sample | 11 +++++++++++ lib/galaxy/config/schemas/config_schema.yml | 15 +++++++++++++++ 3 files changed, 44 insertions(+) diff --git a/doc/source/admin/galaxy_options.rst b/doc/source/admin/galaxy_options.rst index 4608e85d32b..d0e3ce3a5ec 100644 --- a/doc/source/admin/galaxy_options.rst +++ b/doc/source/admin/galaxy_options.rst @@ -5754,6 +5754,24 @@ :Type: int +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +``iwc_manifest_refresh_interval`` +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +:Description: + Time (in seconds) between celery-beat triggered refreshes of the + in-process IWC workflow manifest cache used by the agent-ops + layer. Default matches the cache's in-process TTL so the cache + stays continuously warm rather than expiring between user-driven + hits. Failures are logged and the prior cached copy is retained. + Only registered when ``inference_services`` is configured (i.e. + ChatGXY is in use). Set to 0 to disable automatic refresh -- + agent-ops callers will then fall back to lazy on-demand fetching + with the same hour TTL. Requires celery. +:Default: ``3600`` +:Type: int + + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ ``enable_tool_recommendations`` ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/lib/galaxy/config/sample/galaxy.yml.sample b/lib/galaxy/config/sample/galaxy.yml.sample index dec50bd87a9..cb3502f35a2 100644 --- a/lib/galaxy/config/sample/galaxy.yml.sample +++ b/lib/galaxy/config/sample/galaxy.yml.sample @@ -3120,6 +3120,17 @@ galaxy: # --refresh``. Requires celery. #gtn_database_refresh_interval: 86400 + # Time (in seconds) between celery-beat triggered refreshes of the + # in-process IWC workflow manifest cache used by the agent-ops layer. + # Default matches the cache's in-process TTL so the cache stays + # continuously warm rather than expiring between user-driven hits. + # Failures are logged and the prior cached copy is retained. Only + # registered when ``inference_services`` is configured (i.e. ChatGXY + # is in use). Set to 0 to disable automatic refresh -- agent-ops + # callers will then fall back to lazy on-demand fetching with the same + # hour TTL. Requires celery. + #iwc_manifest_refresh_interval: 3600 + # Allow the display of tool recommendations in workflow editor and # after tool execution. If it is enabled and set to true, please # enable 'tool_recommendation_model_path' as well diff --git a/lib/galaxy/config/schemas/config_schema.yml b/lib/galaxy/config/schemas/config_schema.yml index 4b7e493fd35..a365f988fd8 100644 --- a/lib/galaxy/config/schemas/config_schema.yml +++ b/lib/galaxy/config/schemas/config_schema.yml @@ -4260,6 +4260,21 @@ mapping: can still refresh on demand via ``python -m galaxy.agents.gtn --refresh``. Requires celery. + iwc_manifest_refresh_interval: + type: int + default: 3600 + required: false + desc: | + Time (in seconds) between celery-beat triggered refreshes of the + in-process IWC workflow manifest cache used by the agent-ops layer. + Default matches the cache's in-process TTL so the cache stays + continuously warm rather than expiring between user-driven hits. + Failures are logged and the prior cached copy is retained. Only + registered when ``inference_services`` is configured (i.e. ChatGXY + is in use). Set to 0 to disable automatic refresh -- agent-ops + callers will then fall back to lazy on-demand fetching with the + same hour TTL. Requires celery. + enable_tool_recommendations: type: bool default: false From d273c9d6c8d3d8a3a06e86b6902d8519e06cc694 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 27 May 2026 17:17:54 -0400 Subject: [PATCH 07/13] Pre-warm the IWC workflow manifest cache on a celery beat schedule. The new refresh_manifest helper in iwc.py is a force-fetch sibling of fetch_manifest that holds the cache lock across the network call, so concurrent on-demand readers see either the previous value or the new one but never an empty cache mid-write. The celery task wraps it with the periodic-task error contract (swallow failures, retain prior copy). Registration uses the same inference_services gate as the GTN refresh -- IWC pre-warm only matters when the agent-ops layer is exposed. --- lib/galaxy/agents/iwc.py | 21 ++++++++++++++++++ lib/galaxy/celery/__init__.py | 5 +++++ lib/galaxy/celery/tasks.py | 23 ++++++++++++++++++++ test/unit/agents/test_iwc.py | 41 +++++++++++++++++++++++++++++++++++ test/unit/app/test_celery.py | 16 ++++++++++++-- 5 files changed, 104 insertions(+), 2 deletions(-) diff --git a/lib/galaxy/agents/iwc.py b/lib/galaxy/agents/iwc.py index b65da114178..f3a46016447 100644 --- a/lib/galaxy/agents/iwc.py +++ b/lib/galaxy/agents/iwc.py @@ -53,6 +53,27 @@ def fetch_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: return manifest +def refresh_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: + """Force-fetch the manifest and replace the cached entry. + + Used by the celery-beat pre-warm task. The lock is held across the + network fetch so a concurrent on-demand call from a request handler + sees either the previous value or the new one -- never an empty cache + mid-write. Kept as a separate function from ``fetch_manifest`` because + the error contracts differ: lazy fetch propagates (caller can't + continue without the data); this one is called from a periodic task + that has to tolerate transient failure. + """ + with _manifest_cache_lock: + response = requests.get(IWC_MANIFEST_URL, timeout=timeout) + response.raise_for_status() + manifest = response.json() + if not isinstance(manifest, list): + raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") + _manifest_cache[_CACHE_KEY] = manifest + return manifest + + def all_workflows(manifest: list[dict[str, Any]]) -> list[dict[str, Any]]: """Flatten the manifest into a single list of workflow entries.""" workflows: list[dict[str, Any]] = [] diff --git a/lib/galaxy/celery/__init__.py b/lib/galaxy/celery/__init__.py index 956d4885eed..5759384c97c 100644 --- a/lib/galaxy/celery/__init__.py +++ b/lib/galaxy/celery/__init__.py @@ -294,6 +294,11 @@ def setup_periodic_tasks(config, celery_app): ): schedule_task("refresh_gtn_database", config.gtn_database_refresh_interval) + # IWC manifest pre-warm only matters when the agent-ops layer is exposed -- + # same inference_services gate as the GTN refresh above. + if getattr(config, "inference_services", None) and config.iwc_manifest_refresh_interval: + schedule_task("refresh_iwc_manifest", config.iwc_manifest_refresh_interval) + if config.celery_user_concurrency_limit: # Run cleanup every 5 minutes (300 seconds) schedule_task("cleanup_stale_concurrency_slots", 300) diff --git a/lib/galaxy/celery/tasks.py b/lib/galaxy/celery/tasks.py index 44ec4f7db72..17826ef1d68 100644 --- a/lib/galaxy/celery/tasks.py +++ b/lib/galaxy/celery/tasks.py @@ -830,6 +830,29 @@ def renew_vault_token(vault: Vault): renew_vault_token_if_needed(vault) +@galaxy_task(action="refreshing IWC workflow manifest cache") +def refresh_iwc_manifest(config: GalaxyAppConfiguration): + """Pre-warm the in-process IWC manifest cache. + + The agent-ops layer caches the manifest at module scope with an hour + TTL; without this task the first user-driven IWC call after a worker + restart pays the full network fetch. Failures are logged and swallowed + so an iwc.galaxyproject.org outage doesn't kill the periodic queue -- + on-demand callers still get the prior cached copy until the TTL lapses. + """ + # Local import keeps the iwc module (and its cachetools / requests + # imports) out of celery's startup graph -- they only load on the + # workers that actually run this task. + from galaxy.agents import iwc + + try: + manifest = iwc.refresh_manifest() + except Exception as e: # noqa: BLE001 -- best-effort warm; resilience over precision + log.warning("refresh_iwc_manifest: fetch failed, keeping existing cache: %s", e) + return + log.info("refresh_iwc_manifest: cached %s top-level manifest entries", len(manifest)) + + @galaxy_task(action="refreshing GTN training database") def refresh_gtn_database(config: GalaxyAppConfiguration): """HEAD depot for the GTN search database and re-download only when newer. diff --git a/test/unit/agents/test_iwc.py b/test/unit/agents/test_iwc.py index 337b1cf0e8e..bf716f0f677 100644 --- a/test/unit/agents/test_iwc.py +++ b/test/unit/agents/test_iwc.py @@ -48,6 +48,47 @@ def test_fetch_manifest_caches_response(): assert mock_get.call_count == 1 +def test_refresh_manifest_replaces_cached_value(): + with patch("galaxy.agents.iwc.requests.get") as mock_get: + mock_get.return_value.json.return_value = SAMPLE_MANIFEST + mock_get.return_value.raise_for_status.return_value = None + + first = iwc.refresh_manifest() + assert first == SAMPLE_MANIFEST + + new_manifest = [{"workflows": [{"trsID": "#workflow/x/y/main"}]}] + mock_get.return_value.json.return_value = new_manifest + + second = iwc.refresh_manifest() + assert second == new_manifest + # And the next on-demand fetch sees the refreshed value + assert iwc.fetch_manifest() == new_manifest + + +def test_refresh_manifest_failure_leaves_prior_cache(): + with patch("galaxy.agents.iwc.requests.get") as mock_get: + mock_get.return_value.json.return_value = SAMPLE_MANIFEST + mock_get.return_value.raise_for_status.return_value = None + iwc.fetch_manifest() # prime the cache + + mock_get.side_effect = RuntimeError("boom") + with pytest.raises(RuntimeError): + iwc.refresh_manifest() + + # fetch_manifest still returns the previously-cached value + mock_get.side_effect = None + assert iwc.fetch_manifest() == SAMPLE_MANIFEST + + +def test_refresh_manifest_rejects_non_list_payload(): + with patch("galaxy.agents.iwc.requests.get") as mock_get: + mock_get.return_value.json.return_value = {"not": "a list"} + mock_get.return_value.raise_for_status.return_value = None + + with pytest.raises(ValueError, match="did not return a JSON array"): + iwc.refresh_manifest() + + def test_clean_readme_summary_strips_headers_and_truncates(): body = "First line that has plenty of content. Second line continues the thought. " readme = "# Heading\n\n" + (body * 10) diff --git a/test/unit/app/test_celery.py b/test/unit/app/test_celery.py index 48dbc02ea6d..bf2bcb1d040 100644 --- a/test/unit/app/test_celery.py +++ b/test/unit/app/test_celery.py @@ -27,9 +27,10 @@ def test_default_configuration(): "task": "galaxy.cleanup_short_term_storage", "schedule": galaxy_conf.short_term_storage_cleanup_interval, } - # GTN refresh is gated on inference_services being configured; default - # config doesn't set it, so the schedule isn't registered here. + # GTN and IWC refreshes are gated on inference_services being configured; + # default config doesn't set it, so neither schedule is registered here. assert "refresh-gtn-database" not in conf.beat_schedule + assert "refresh-iwc-manifest" not in conf.beat_schedule def test_gtn_refresh_schedules_when_inference_configured(): @@ -43,6 +44,17 @@ def test_gtn_refresh_schedules_when_inference_configured(): } +def test_iwc_refresh_schedules_when_inference_configured(): + config = GalaxyAppConfiguration(override_tempdir=False) + config.inference_services = {"default": {"model": "test"}} + app = GalaxyCelery("test-iwc-schedule") + setup_periodic_tasks(config, app) + assert app.conf.beat_schedule["refresh-iwc-manifest"] == { + "task": "galaxy.refresh_iwc_manifest", + "schedule": config.iwc_manifest_refresh_interval, + } + + def test_galaxycelery_trim_module_name(): gc = GalaxyCelery() assert gc.trim_module_name("notgalaxy.celery.tasks") == "notgalaxy.celery.tasks" From 1b0d08bd5b4fac803e334138d62c11a87c308cdf Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Fri, 29 May 2026 07:33:25 -0400 Subject: [PATCH 08/13] Fetch the IWC manifest outside the cache lock refresh_manifest held _manifest_cache_lock across the network fetch, so a slow or hanging iwc.galaxyproject.org would block on-demand fetch_manifest readers (the agent-ops request handlers) for up to the full timeout even though a good cached copy was sitting right there. Move the GET and validation outside the lock and only hold it for the single-assignment write. A concurrent reader still sees either the previous value or the new one -- never an empty cache mid-write -- but no longer waits behind the beat task's network round-trip. --- lib/galaxy/agents/iwc.py | 22 ++++++++++++---------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/lib/galaxy/agents/iwc.py b/lib/galaxy/agents/iwc.py index f3a46016447..14a1c6e7b12 100644 --- a/lib/galaxy/agents/iwc.py +++ b/lib/galaxy/agents/iwc.py @@ -56,22 +56,24 @@ def fetch_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: def refresh_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: """Force-fetch the manifest and replace the cached entry. - Used by the celery-beat pre-warm task. The lock is held across the - network fetch so a concurrent on-demand call from a request handler - sees either the previous value or the new one -- never an empty cache - mid-write. Kept as a separate function from ``fetch_manifest`` because + Used by the celery-beat pre-warm task. The network fetch runs outside + the cache lock so a slow or hanging iwc.galaxyproject.org doesn't block + on-demand ``fetch_manifest`` readers -- they keep getting the previous + cached copy. Only the single-assignment cache write is locked, so a + concurrent reader sees either the previous value or the new one, never + an empty cache mid-write. Kept separate from ``fetch_manifest`` because the error contracts differ: lazy fetch propagates (caller can't continue without the data); this one is called from a periodic task that has to tolerate transient failure. """ + response = requests.get(IWC_MANIFEST_URL, timeout=timeout) + response.raise_for_status() + manifest = response.json() + if not isinstance(manifest, list): + raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") with _manifest_cache_lock: - response = requests.get(IWC_MANIFEST_URL, timeout=timeout) - response.raise_for_status() - manifest = response.json() - if not isinstance(manifest, list): - raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") _manifest_cache[_CACHE_KEY] = manifest - return manifest + return manifest def all_workflows(manifest: list[dict[str, Any]]) -> list[dict[str, Any]]: From d053bfae0b6c02d2217013af81c11768686586d8 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Fri, 29 May 2026 07:41:04 -0400 Subject: [PATCH 09/13] Use the GalaxyAI name in the GTN/IWC refresh config and comments The agent assistant was rebranded from ChatGXY to GalaxyAI, but this branch's new config descriptions and code comments still said ChatGXY. Swap them over and regenerate the sample yml + options rst from the schema. The lowercase depot bucket path in gtn_database_url stays as-is -- that's a real URL, not a brand reference. --- doc/source/admin/galaxy_options.rst | 6 +++--- lib/galaxy/celery/__init__.py | 2 +- lib/galaxy/celery/tasks.py | 2 +- lib/galaxy/config/sample/galaxy.yml.sample | 4 ++-- lib/galaxy/config/schemas/config_schema.yml | 4 ++-- 5 files changed, 9 insertions(+), 9 deletions(-) diff --git a/doc/source/admin/galaxy_options.rst b/doc/source/admin/galaxy_options.rst index d0e3ce3a5ec..dbb525bed4b 100644 --- a/doc/source/admin/galaxy_options.rst +++ b/doc/source/admin/galaxy_options.rst @@ -5747,8 +5747,8 @@ ``gtn_database_path``; live handlers pick up the new copy on their next query since GTNSearchDB opens a read-only connection per call. Only registered when ``inference_services`` is configured - (i.e. ChatGXY is in use). Set to 0 to disable automatic refresh -- - admins can still refresh on demand via ``python -m + (i.e. GalaxyAI is in use). Set to 0 to disable automatic refresh + -- admins can still refresh on demand via ``python -m galaxy.agents.gtn --refresh``. Requires celery. :Default: ``86400`` :Type: int @@ -5765,7 +5765,7 @@ stays continuously warm rather than expiring between user-driven hits. Failures are logged and the prior cached copy is retained. Only registered when ``inference_services`` is configured (i.e. - ChatGXY is in use). Set to 0 to disable automatic refresh -- + GalaxyAI is in use). Set to 0 to disable automatic refresh -- agent-ops callers will then fall back to lazy on-demand fetching with the same hour TTL. Requires celery. :Default: ``3600`` diff --git a/lib/galaxy/celery/__init__.py b/lib/galaxy/celery/__init__.py index 5759384c97c..d6501248e98 100644 --- a/lib/galaxy/celery/__init__.py +++ b/lib/galaxy/celery/__init__.py @@ -283,7 +283,7 @@ def setup_periodic_tasks(config, celery_app): if config.vault_token_renewal_interval: schedule_task("renew_vault_token", config.vault_token_renewal_interval) - # Only schedule if ChatGXY infrastructure is configured -- the GTN database + # Only schedule if GalaxyAI infrastructure is configured -- the GTN database # serves the gtn_training agent, which only functions when inference_services # is set up. Without this gate every Galaxy install would pull from depot # on the default interval, even ones that don't use the agent. diff --git a/lib/galaxy/celery/tasks.py b/lib/galaxy/celery/tasks.py index 17826ef1d68..f2be0cd4565 100644 --- a/lib/galaxy/celery/tasks.py +++ b/lib/galaxy/celery/tasks.py @@ -858,7 +858,7 @@ def refresh_gtn_database(config: GalaxyAppConfiguration): """HEAD depot for the GTN search database and re-download only when newer. Handlers open the database read-only per query, so an atomic rename here - is picked up by the next ChatGXY request without a restart. The HEAD-first + is picked up by the next GalaxyAI request without a restart. The HEAD-first pattern keeps the steady-state cost to a few hundred bytes per tick -- only when depot has actually been updated do we pull the full ~17MB. Failures are logged and swallowed so a depot outage doesn't kill the diff --git a/lib/galaxy/config/sample/galaxy.yml.sample b/lib/galaxy/config/sample/galaxy.yml.sample index cb3502f35a2..c94c2358253 100644 --- a/lib/galaxy/config/sample/galaxy.yml.sample +++ b/lib/galaxy/config/sample/galaxy.yml.sample @@ -3115,7 +3115,7 @@ galaxy: # ``gtn_database_path``; live handlers pick up the new copy on their # next query since GTNSearchDB opens a read-only connection per call. # Only registered when ``inference_services`` is configured (i.e. - # ChatGXY is in use). Set to 0 to disable automatic refresh -- admins + # GalaxyAI is in use). Set to 0 to disable automatic refresh -- admins # can still refresh on demand via ``python -m galaxy.agents.gtn # --refresh``. Requires celery. #gtn_database_refresh_interval: 86400 @@ -3125,7 +3125,7 @@ galaxy: # Default matches the cache's in-process TTL so the cache stays # continuously warm rather than expiring between user-driven hits. # Failures are logged and the prior cached copy is retained. Only - # registered when ``inference_services`` is configured (i.e. ChatGXY + # registered when ``inference_services`` is configured (i.e. GalaxyAI # is in use). Set to 0 to disable automatic refresh -- agent-ops # callers will then fall back to lazy on-demand fetching with the same # hour TTL. Requires celery. diff --git a/lib/galaxy/config/schemas/config_schema.yml b/lib/galaxy/config/schemas/config_schema.yml index a365f988fd8..f737f1072c1 100644 --- a/lib/galaxy/config/schemas/config_schema.yml +++ b/lib/galaxy/config/schemas/config_schema.yml @@ -4256,7 +4256,7 @@ mapping: ``gtn_database_path``; live handlers pick up the new copy on their next query since GTNSearchDB opens a read-only connection per call. Only registered when ``inference_services`` is configured (i.e. - ChatGXY is in use). Set to 0 to disable automatic refresh -- admins + GalaxyAI is in use). Set to 0 to disable automatic refresh -- admins can still refresh on demand via ``python -m galaxy.agents.gtn --refresh``. Requires celery. @@ -4270,7 +4270,7 @@ mapping: Default matches the cache's in-process TTL so the cache stays continuously warm rather than expiring between user-driven hits. Failures are logged and the prior cached copy is retained. Only - registered when ``inference_services`` is configured (i.e. ChatGXY + registered when ``inference_services`` is configured (i.e. GalaxyAI is in use). Set to 0 to disable automatic refresh -- agent-ops callers will then fall back to lazy on-demand fetching with the same hour TTL. Requires celery. From cb6d67dd370f64cd7907412e077ff8ab6e09abc3 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Sun, 31 May 2026 13:50:35 -0400 Subject: [PATCH 10/13] Extract _download_manifest to dedupe the IWC fetch block fetch_manifest and refresh_manifest both had a byte-identical fetch/raise_for_status/json/isinstance block. Pull it into a private helper so the two functions differ only in their locking strategy -- which is the actual point refresh_manifest's docstring labors to explain. --- lib/galaxy/agents/iwc.py | 22 ++++++++++++---------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/lib/galaxy/agents/iwc.py b/lib/galaxy/agents/iwc.py index 14a1c6e7b12..cc5386aa77f 100644 --- a/lib/galaxy/agents/iwc.py +++ b/lib/galaxy/agents/iwc.py @@ -33,6 +33,16 @@ def clear_manifest_cache() -> None: _manifest_cache.clear() +def _download_manifest(timeout: float) -> list[dict[str, Any]]: + """Fetch and validate the IWC manifest over the network, bypassing the cache.""" + response = requests.get(IWC_MANIFEST_URL, timeout=timeout) + response.raise_for_status() + manifest = response.json() + if not isinstance(manifest, list): + raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") + return manifest + + def fetch_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: """Fetch the IWC manifest, returning a cached copy when fresh. @@ -44,11 +54,7 @@ def fetch_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: if cached is not None: return cached - response = requests.get(IWC_MANIFEST_URL, timeout=timeout) - response.raise_for_status() - manifest = response.json() - if not isinstance(manifest, list): - raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") + manifest = _download_manifest(timeout) _manifest_cache[_CACHE_KEY] = manifest return manifest @@ -66,11 +72,7 @@ def refresh_manifest(timeout: float = 30.0) -> list[dict[str, Any]]: continue without the data); this one is called from a periodic task that has to tolerate transient failure. """ - response = requests.get(IWC_MANIFEST_URL, timeout=timeout) - response.raise_for_status() - manifest = response.json() - if not isinstance(manifest, list): - raise ValueError(f"IWC manifest at {IWC_MANIFEST_URL} did not return a JSON array") + manifest = _download_manifest(timeout) with _manifest_cache_lock: _manifest_cache[_CACHE_KEY] = manifest return manifest From 0064de80cffdf81ccfd0984671c98756a69715b1 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Sun, 31 May 2026 13:50:37 -0400 Subject: [PATCH 11/13] Hoist the iwc import in refresh_iwc_manifest to module top The local import claimed to keep cachetools/requests out of celery's startup graph, but cachetools is already pulled in transitively (NotificationManager -> sse_dispatch) and requests is resident in any worker that boots galaxy.app. There's no circular-import risk either, and the sibling refresh_gtn_database imports GTNSearchDB at the top -- so hoist this one for consistency and drop the comment. --- lib/galaxy/celery/tasks.py | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/lib/galaxy/celery/tasks.py b/lib/galaxy/celery/tasks.py index f2be0cd4565..48e079decd6 100644 --- a/lib/galaxy/celery/tasks.py +++ b/lib/galaxy/celery/tasks.py @@ -26,6 +26,7 @@ from sqlalchemy import ( ) from galaxy import model +from galaxy.agents import iwc from galaxy.agents.gtn import GTNSearchDB from galaxy.celery import ( celery_app, @@ -840,11 +841,6 @@ def refresh_iwc_manifest(config: GalaxyAppConfiguration): so an iwc.galaxyproject.org outage doesn't kill the periodic queue -- on-demand callers still get the prior cached copy until the TTL lapses. """ - # Local import keeps the iwc module (and its cachetools / requests - # imports) out of celery's startup graph -- they only load on the - # workers that actually run this task. - from galaxy.agents import iwc - try: manifest = iwc.refresh_manifest() except Exception as e: # noqa: BLE001 -- best-effort warm; resilience over precision From dfe8d17bd1a32f08fc5ef57125a93f1b820ea11e Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Sun, 31 May 2026 17:36:22 -0400 Subject: [PATCH 12/13] Quiet mypy on the inference_services test assignment inference_services is a schema-loaded config option, so it's not a static attribute on GalaxyAppConfiguration -- production code reads it via getattr for the same reason. The two celery schedule tests set it directly, which trips mypy's attr-defined check, so mark those assignments type: ignore the way the other config-poking unit tests (test_vault, test_error_reporting) do. --- test/unit/app/test_celery.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/test/unit/app/test_celery.py b/test/unit/app/test_celery.py index bf2bcb1d040..fb68368560d 100644 --- a/test/unit/app/test_celery.py +++ b/test/unit/app/test_celery.py @@ -35,7 +35,7 @@ def test_default_configuration(): def test_gtn_refresh_schedules_when_inference_configured(): config = GalaxyAppConfiguration(override_tempdir=False) - config.inference_services = {"default": {"model": "test"}} + config.inference_services = {"default": {"model": "test"}} # type: ignore[attr-defined] app = GalaxyCelery("test-gtn-schedule") setup_periodic_tasks(config, app) assert app.conf.beat_schedule["refresh-gtn-database"] == { @@ -46,7 +46,7 @@ def test_gtn_refresh_schedules_when_inference_configured(): def test_iwc_refresh_schedules_when_inference_configured(): config = GalaxyAppConfiguration(override_tempdir=False) - config.inference_services = {"default": {"model": "test"}} + config.inference_services = {"default": {"model": "test"}} # type: ignore[attr-defined] app = GalaxyCelery("test-iwc-schedule") setup_periodic_tasks(config, app) assert app.conf.beat_schedule["refresh-iwc-manifest"] == { From d83c8e102e93f3df1e5e9dd84c23b92d58836cda Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Sat, 6 Jun 2026 13:23:11 +0200 Subject: [PATCH 13/13] Drop overly defensive getattre --- lib/galaxy/celery/__init__.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/lib/galaxy/celery/__init__.py b/lib/galaxy/celery/__init__.py index d6501248e98..a251ebadd22 100644 --- a/lib/galaxy/celery/__init__.py +++ b/lib/galaxy/celery/__init__.py @@ -287,16 +287,12 @@ def setup_periodic_tasks(config, celery_app): # serves the gtn_training agent, which only functions when inference_services # is set up. Without this gate every Galaxy install would pull from depot # on the default interval, even ones that don't use the agent. - if ( - getattr(config, "inference_services", None) - and config.gtn_database_refresh_interval - and config.gtn_database_path - ): + if config.inference_services and config.gtn_database_refresh_interval and config.gtn_database_path: schedule_task("refresh_gtn_database", config.gtn_database_refresh_interval) # IWC manifest pre-warm only matters when the agent-ops layer is exposed -- # same inference_services gate as the GTN refresh above. - if getattr(config, "inference_services", None) and config.iwc_manifest_refresh_interval: + if config.inference_services and config.iwc_manifest_refresh_interval: schedule_task("refresh_iwc_manifest", config.iwc_manifest_refresh_interval) if config.celery_user_concurrency_limit: