diff --git a/client/src/composables/zipExplorer.ts b/client/src/composables/zipExplorer.ts index e88c844f00e..d5846827a66 100644 --- a/client/src/composables/zipExplorer.ts +++ b/client/src/composables/zipExplorer.ts @@ -310,7 +310,7 @@ export function validateLocalZipFile(file?: File | null): string { } export function isLocalZipFile(file?: File | null): boolean { - return Boolean(file) && file?.type === "application/zip"; + return Boolean(file) && (file?.type === "application/zip" || file?.type === "application/x-zip-compressed"); } export async function isRemoteZipFile(url: string): Promise { diff --git a/lib/galaxy/config/sample/datatypes_conf.xml.sample b/lib/galaxy/config/sample/datatypes_conf.xml.sample index 7fd4b7a2a5a..8309118a83a 100644 --- a/lib/galaxy/config/sample/datatypes_conf.xml.sample +++ b/lib/galaxy/config/sample/datatypes_conf.xml.sample @@ -150,6 +150,7 @@ + @@ -1180,6 +1181,7 @@ + @@ -1395,6 +1397,7 @@ + diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index dcbfeecda92..bc8da6d4797 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -4872,3 +4872,102 @@ class Hic(Binary): with open(dataset.get_file_name(), "rb") as handle: header_bytes = handle.read(8) dataset.metadata.version = struct.unpack(" bool: + """ + Determining if the file is in safetensors format + >>> from galaxy.datatypes.sniff import get_test_fname + >>> fname = get_test_fname('cellpose_model_safetensors.safetensors') + >>> Safetensors().sniff(fname) + True + >>> fname = get_test_fname('test_charmm.vel') + >>> Safetensors().sniff(fname) + False + """ + try: + # Safetensors files start with an 8-byte little-endian integer + # indicating the size of the JSON header + if len(file_prefix.contents_header_bytes) < 8: + return False + + header_size = int.from_bytes(file_prefix.contents_header_bytes[:8], "little") + + # Currently, there's a limit on the size of the header of 100MB to prevent parsing extremely large JSON headers + # In practice, safetensors headers are typically just a few KB to MB + # (containing tensor names, shapes, dtypes, and offsets - rarely exceeds 1-10MB even for large models) + # But in theory it is possible to have 100 MB header + # more info here: https://github.com/huggingface/safetensors?tab=readme-ov-file#benefits + if header_size == 0 or header_size > 10**8: # 100MB max for JSON header + return False + + # Check if file is large enough to contain the full header + if file_prefix.file_size < 8 + header_size: + return False + + # CRITICAL: Check if header begins with '{' character (0x7B) as per safetensors spec + # This is required by the format and helps distinguish from other binary formats + # Only check 1 byte to avoid issues with malicious header_size values + # more info here: https://github.com/huggingface/safetensors?tab=readme-ov-file#format + if file_prefix.contents_header_bytes[8] != 0x7B: + return False + + # Check if header ends with '}' character (0x7D) as per safetensors spec + # This requires reading more data if header extends beyond the prefix + header_end_pos = 8 + header_size - 1 + if header_end_pos < len(file_prefix.contents_header_bytes): + # Header end is within the prefix + if file_prefix.contents_header_bytes[header_end_pos] != 0x7D: + return False + else: + # Header extends beyond prefix, need to check from file + with open(file_prefix.filename, "rb") as f: + f.seek(header_end_pos) + last_header_byte = f.read(1) + if len(last_header_byte) != 1 or last_header_byte[0] != 0x7D: + return False + + # Read the full header for JSON parsing + if 8 + header_size <= len(file_prefix.contents_header_bytes): + # Entire header is in the prefix + header_bytes = file_prefix.contents_header_bytes[8 : 8 + header_size] + else: + # Need to read full header from file + with open(file_prefix.filename, "rb") as f: + f.seek(8) + header_bytes = f.read(header_size) + + if len(header_bytes) != header_size: + return False + + # Parse the validated JSON header + header = json.loads(header_bytes.decode("utf-8")) + # check if header is a dict + if not isinstance(header, dict): + return False + # Basic validation: check if it looks like safetensors metadata + # Safetensors headers should have entries with data_offsets + has_valid_entries = False + for key, value in header.items(): + if key == "__metadata__": # Special metadata key + continue + if isinstance(value, dict) and "data_offsets" in value: + has_valid_entries = True + break + + return has_valid_entries + + except Exception: + # Any exception during parsing means it's not a valid safetensors file + return False diff --git a/lib/galaxy/datatypes/test/1.auspicejson b/lib/galaxy/datatypes/test/1.auspicejson new file mode 100644 index 00000000000..541dcdaefb2 --- /dev/null +++ b/lib/galaxy/datatypes/test/1.auspicejson @@ -0,0 +1,14 @@ +{ + "version": "v2", + "meta": { + "title": "Minimal AuspiceJSON", + "updated": "2025-02-05", + "panels": ["tree"] + }, + "tree": { + "name": "1", + "node_attrs": { + "div": 1 + } + } +} diff --git a/lib/galaxy/datatypes/test/cellpose_model_safetensors.safetensors b/lib/galaxy/datatypes/test/cellpose_model_safetensors.safetensors new file mode 100644 index 00000000000..d24e51dc49a Binary files /dev/null and b/lib/galaxy/datatypes/test/cellpose_model_safetensors.safetensors differ diff --git a/lib/galaxy/datatypes/text.py b/lib/galaxy/datatypes/text.py index f990f033815..f002e8cce64 100644 --- a/lib/galaxy/datatypes/text.py +++ b/lib/galaxy/datatypes/text.py @@ -738,6 +738,57 @@ class VitessceJson(Json): return False +@build_sniff_from_prefix +class AuspiceJson(Json): + """ + Auspice is a visualization tool for phylogenetic trees and associated data. + It uses JSON format to represent the tree structure and metadata. + """ + + file_ext = "auspice.json" + + def set_peek(self, dataset: DatasetProtocol, **kwd) -> None: + super().set_peek(dataset) + if not dataset.dataset.purged: + dataset.blurb = "AuspiceJSON" + + def sniff_prefix(self, file_prefix: FilePrefix) -> bool: + """ + Determines whether the file is in Auspice v2 JSON by looking for keys + like "version", "meta" and "updated" that are both required by the + https://docs.nextstrain.org/projects/auspice/en/stable/releases/v2.html format + and also will be in the first part of the file + + >>> from galaxy.datatypes.sniff import get_test_fname + >>> fname = get_test_fname( '1.json' ) + >>> AuspiceJson().sniff( fname ) + False + >>> fname = get_test_fname( '1.auspicejson' ) + >>> AuspiceJson().sniff( fname ) + True + """ + is_auspicejson = False + if self._looks_like_json(file_prefix): + is_auspicejson = self._looks_like_is_auspicejson(file_prefix) + return is_auspicejson + + def _looks_like_is_auspicejson(self, file_prefix: FilePrefix, load_size: int = 20000) -> bool: + """ + Expects JSON to start with { and 'meta', 'tree', 'updated' and 'nodes' to be present as keys in the JSON structure. + """ + try: + with open(file_prefix.filename) as fh: + segment_str = fh.read(load_size) + + if segment_str.startswith("{") and all( + x in segment_str for x in ["version", "meta", "updated", "panels"] + ): + return True + except Exception: + pass + return False + + @build_sniff_from_prefix class Obo(Text): """