mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #16069 from jmchilton/chunking_tests
Dataset chunking tests (and small fixes)
This commit is contained in:
@@ -8531,11 +8531,15 @@ export interface operations {
|
||||
/** @description If non-null, get the specified filename from the extra files for this dataset. */
|
||||
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
|
||||
/** @description The query parameter 'raw' should be considered experimental and may be dropped at some point in the future without warning. Generally, data should be processed by its datatype prior to display. */
|
||||
/** @description Set this for datatypes that allow chunked display through the display_data method to enable chunking. This specifies a byte offset into the target dataset's display. */
|
||||
/** @description If offset is set, this recommends 'how large' the next chunk should be. This is not respected or interpreted uniformly and should be interpreted as a very loose recommendation. Different datatypes interpret 'largeness' differently - for bam datasets this is a number of lines whereas for tabular datatypes this is interpreted as a number of bytes. */
|
||||
query?: {
|
||||
preview?: boolean;
|
||||
filename?: string;
|
||||
to_ext?: string;
|
||||
raw?: boolean;
|
||||
offset?: number;
|
||||
ck_size?: number;
|
||||
};
|
||||
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
|
||||
header?: {
|
||||
@@ -8567,11 +8571,15 @@ export interface operations {
|
||||
/** @description If non-null, get the specified filename from the extra files for this dataset. */
|
||||
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
|
||||
/** @description The query parameter 'raw' should be considered experimental and may be dropped at some point in the future without warning. Generally, data should be processed by its datatype prior to display. */
|
||||
/** @description Set this for datatypes that allow chunked display through the display_data method to enable chunking. This specifies a byte offset into the target dataset's display. */
|
||||
/** @description If offset is set, this recommends 'how large' the next chunk should be. This is not respected or interpreted uniformly and should be interpreted as a very loose recommendation. Different datatypes interpret 'largeness' differently - for bam datasets this is a number of lines whereas for tabular datatypes this is interpreted as a number of bytes. */
|
||||
query?: {
|
||||
preview?: boolean;
|
||||
filename?: string;
|
||||
to_ext?: string;
|
||||
raw?: boolean;
|
||||
offset?: number;
|
||||
ck_size?: number;
|
||||
};
|
||||
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
|
||||
header?: {
|
||||
@@ -10705,11 +10713,15 @@ export interface operations {
|
||||
/** @description If non-null, get the specified filename from the extra files for this dataset. */
|
||||
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
|
||||
/** @description The query parameter 'raw' should be considered experimental and may be dropped at some point in the future without warning. Generally, data should be processed by its datatype prior to display. */
|
||||
/** @description Set this for datatypes that allow chunked display through the display_data method to enable chunking. This specifies a byte offset into the target dataset's display. */
|
||||
/** @description If offset is set, this recommends 'how large' the next chunk should be. This is not respected or interpreted uniformly and should be interpreted as a very loose recommendation. Different datatypes interpret 'largeness' differently - for bam datasets this is a number of lines whereas for tabular datatypes this is interpreted as a number of bytes. */
|
||||
query?: {
|
||||
preview?: boolean;
|
||||
filename?: string;
|
||||
to_ext?: string;
|
||||
raw?: boolean;
|
||||
offset?: number;
|
||||
ck_size?: number;
|
||||
};
|
||||
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
|
||||
header?: {
|
||||
@@ -10743,11 +10755,15 @@ export interface operations {
|
||||
/** @description If non-null, get the specified filename from the extra files for this dataset. */
|
||||
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
|
||||
/** @description The query parameter 'raw' should be considered experimental and may be dropped at some point in the future without warning. Generally, data should be processed by its datatype prior to display. */
|
||||
/** @description Set this for datatypes that allow chunked display through the display_data method to enable chunking. This specifies a byte offset into the target dataset's display. */
|
||||
/** @description If offset is set, this recommends 'how large' the next chunk should be. This is not respected or interpreted uniformly and should be interpreted as a very loose recommendation. Different datatypes interpret 'largeness' differently - for bam datasets this is a number of lines whereas for tabular datatypes this is interpreted as a number of bytes. */
|
||||
query?: {
|
||||
preview?: boolean;
|
||||
filename?: string;
|
||||
to_ext?: string;
|
||||
raw?: boolean;
|
||||
offset?: number;
|
||||
ck_size?: number;
|
||||
};
|
||||
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
|
||||
header?: {
|
||||
|
||||
@@ -617,7 +617,8 @@ class BamNative(CompressedArchive, _BamOrSam):
|
||||
if not offset == -1:
|
||||
try:
|
||||
with pysam.AlignmentFile(dataset.file_name, "rb", check_sq=False) as bamfile:
|
||||
ck_size = 300 # 300 lines
|
||||
if ck_size is None:
|
||||
ck_size = 300 # 300 lines
|
||||
if offset == 0:
|
||||
offset = bamfile.tell()
|
||||
ck_lines = bamfile.text.strip().replace("\t", " ").splitlines() # type: ignore[attr-defined]
|
||||
@@ -627,7 +628,7 @@ class BamNative(CompressedArchive, _BamOrSam):
|
||||
for line_number, alignment in enumerate(bamfile, len(ck_lines)):
|
||||
# return only Header lines if 'header_line_count' exceeds 'ck_size'
|
||||
# FIXME: Can be problematic if bam has million lines of header
|
||||
if line_number > ck_size:
|
||||
if line_number >= ck_size:
|
||||
break
|
||||
|
||||
offset = bamfile.tell()
|
||||
|
||||
@@ -5,6 +5,7 @@ import logging
|
||||
from io import (
|
||||
BytesIO,
|
||||
IOBase,
|
||||
StringIO,
|
||||
)
|
||||
from typing import (
|
||||
Any,
|
||||
@@ -108,6 +109,24 @@ RawQueryParam = Query(
|
||||
),
|
||||
)
|
||||
|
||||
DisplayOffsetQueryParam = Query(
|
||||
default=None,
|
||||
description=(
|
||||
"Set this for datatypes that allow chunked display through the display_data method to enable "
|
||||
"chunking. This specifies a byte offset into the target dataset's display."
|
||||
),
|
||||
)
|
||||
|
||||
DisplayChunkSizeQueryParam = Query(
|
||||
default=None,
|
||||
description=(
|
||||
"If offset is set, this recommends 'how large' the next chunk should be. "
|
||||
"This is not respected or interpreted uniformly and should be interpreted as a very loose recommendation. "
|
||||
"Different datatypes interpret 'largeness' differently - for bam datasets this is a number of lines whereas "
|
||||
"for tabular datatypes this is interpreted as a number of bytes. "
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@router.cbv
|
||||
class FastAPIDatasets:
|
||||
@@ -258,9 +277,11 @@ class FastAPIDatasets:
|
||||
filename: Optional[str] = FilenameQueryParam,
|
||||
to_ext: Optional[str] = ToExtQueryParam,
|
||||
raw: bool = RawQueryParam,
|
||||
offset: Optional[int] = DisplayOffsetQueryParam,
|
||||
ck_size: Optional[int] = DisplayChunkSizeQueryParam,
|
||||
):
|
||||
"""Streams the dataset for download or the contents preview to be displayed in a browser."""
|
||||
return self._display(request, trans, history_content_id, preview, filename, to_ext, raw)
|
||||
return self._display(request, trans, history_content_id, preview, filename, to_ext, raw, offset, ck_size)
|
||||
|
||||
@router.get(
|
||||
"/api/datasets/{history_content_id}/display",
|
||||
@@ -280,9 +301,11 @@ class FastAPIDatasets:
|
||||
filename: Optional[str] = FilenameQueryParam,
|
||||
to_ext: Optional[str] = ToExtQueryParam,
|
||||
raw: bool = RawQueryParam,
|
||||
offset: Optional[int] = DisplayOffsetQueryParam,
|
||||
ck_size: Optional[int] = DisplayChunkSizeQueryParam,
|
||||
):
|
||||
"""Streams the dataset for download or the contents preview to be displayed in a browser."""
|
||||
return self._display(request, trans, history_content_id, preview, filename, to_ext, raw)
|
||||
return self._display(request, trans, history_content_id, preview, filename, to_ext, raw, offset, ck_size)
|
||||
|
||||
def _display(
|
||||
self,
|
||||
@@ -293,12 +316,22 @@ class FastAPIDatasets:
|
||||
filename: Optional[str],
|
||||
to_ext: Optional[str],
|
||||
raw: bool,
|
||||
offset: Optional[int] = None,
|
||||
ck_size: Optional[int] = None,
|
||||
):
|
||||
extra_params = get_query_parameters_from_request_excluding(
|
||||
request, {"preview", "filename", "to_ext", "raw", "dataset"}
|
||||
request, {"preview", "filename", "to_ext", "raw", "dataset", "ck_size", "offset"}
|
||||
)
|
||||
display_data, headers = self.service.display(
|
||||
trans, history_content_id, preview=preview, filename=filename, to_ext=to_ext, raw=raw, **extra_params
|
||||
trans,
|
||||
history_content_id,
|
||||
preview=preview,
|
||||
filename=filename,
|
||||
to_ext=to_ext,
|
||||
raw=raw,
|
||||
offset=offset,
|
||||
ck_size=ck_size,
|
||||
**extra_params,
|
||||
)
|
||||
if isinstance(display_data, IOBase):
|
||||
file_name = getattr(display_data, "name", None)
|
||||
@@ -308,6 +341,8 @@ class FastAPIDatasets:
|
||||
return StreamingResponse(display_data.response(), headers=headers)
|
||||
elif isinstance(display_data, bytes):
|
||||
return StreamingResponse(BytesIO(display_data), headers=headers)
|
||||
elif isinstance(display_data, str):
|
||||
return StreamingResponse(content=StringIO(display_data), headers=headers)
|
||||
return StreamingResponse(display_data, headers=headers)
|
||||
|
||||
@router.get(
|
||||
|
||||
@@ -565,6 +565,8 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin):
|
||||
filename: Optional[str] = None,
|
||||
to_ext: Optional[str] = None,
|
||||
raw: bool = False,
|
||||
offset: Optional[int] = None,
|
||||
ck_size: Optional[int] = None,
|
||||
**kwd,
|
||||
):
|
||||
"""
|
||||
@@ -572,7 +574,7 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin):
|
||||
|
||||
The query parameter 'raw' should be considered experimental and may be dropped at
|
||||
some point in the future without warning. Generally, data should be processed by its
|
||||
datatype prior to display (the defult if raw is unspecified or explicitly false.
|
||||
datatype prior to display (the default if raw is unspecified or explicitly false.
|
||||
"""
|
||||
headers = {}
|
||||
rval: Any = ""
|
||||
@@ -589,6 +591,10 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin):
|
||||
file_path = dataset_instance.file_name
|
||||
rval = open(file_path, "rb")
|
||||
else:
|
||||
if offset is not None:
|
||||
kwd["offset"] = offset
|
||||
if ck_size is not None:
|
||||
kwd["ck_size"] = ck_size
|
||||
rval, headers = dataset_instance.datatype.display_data(
|
||||
trans, dataset_instance, preview, filename, to_ext, **kwd
|
||||
)
|
||||
|
||||
@@ -341,6 +341,118 @@ class TestDatasetsApi(ApiTestCase):
|
||||
self._assert_status_code_is(display_response, 200)
|
||||
assert display_response.text == contents
|
||||
|
||||
def test_dataprovider_chunk(self, history_id):
|
||||
contents = textwrap.dedent(
|
||||
"""\
|
||||
1 2 3 4
|
||||
A B C D
|
||||
10 20 30 40
|
||||
"""
|
||||
)
|
||||
# test first chunk
|
||||
hda1 = self.dataset_populator.new_dataset(history_id, content=contents, wait=True)
|
||||
kwds = {
|
||||
"data_type": "raw_data",
|
||||
"provider": "chunk",
|
||||
"chunk_index": "0",
|
||||
"chunk_size": "5",
|
||||
}
|
||||
|
||||
display_response = self._get(f"datasets/{hda1['id']}", kwds)
|
||||
self._assert_status_code_is(display_response, 200)
|
||||
display = display_response.json()
|
||||
self._assert_has_key(display, "data")
|
||||
assert display["data"] == ["1 2"]
|
||||
|
||||
# test index
|
||||
kwds = {
|
||||
"data_type": "raw_data",
|
||||
"provider": "chunk",
|
||||
"chunk_index": "1",
|
||||
"chunk_size": "5",
|
||||
}
|
||||
|
||||
display_response = self._get(f"datasets/{hda1['id']}", kwds)
|
||||
self._assert_status_code_is(display_response, 200)
|
||||
display = display_response.json()
|
||||
self._assert_has_key(display, "data")
|
||||
assert display["data"] == [" 3 "]
|
||||
|
||||
# test line breaks
|
||||
kwds = {
|
||||
"data_type": "raw_data",
|
||||
"provider": "chunk",
|
||||
"chunk_index": "0",
|
||||
"chunk_size": "20",
|
||||
}
|
||||
|
||||
display_response = self._get(f"datasets/{hda1['id']}", kwds)
|
||||
self._assert_status_code_is(display_response, 200)
|
||||
display = display_response.json()
|
||||
self._assert_has_key(display, "data")
|
||||
assert "\nA" in display["data"][0]
|
||||
|
||||
def test_bam_chunking_through_display_endpoint(self, history_id):
|
||||
# This endpoint does not use data providers and instead overrides display_data
|
||||
# in the bam datatype. This is the endpoint is very close to the legacy non-API
|
||||
# controller endpoint used by the UI to produce these chunks.
|
||||
bam_dataset = self.dataset_populator.new_bam_dataset(history_id, self.test_data_resolver)
|
||||
bam_id = bam_dataset["id"]
|
||||
|
||||
chunk_1 = self._display_chunk(bam_id, 0, 1)
|
||||
self._assert_has_keys(chunk_1, "offset", "ck_data")
|
||||
|
||||
offset = chunk_1["offset"]
|
||||
|
||||
chunk_2 = self._display_chunk(bam_id, offset, 1)
|
||||
assert chunk_2["offset"] > offset
|
||||
# chunk_1 just contains all the headers so this check wouldn't work.
|
||||
assert len(chunk_2["ck_data"].split("\n")) == 1
|
||||
|
||||
double_chunk = self._display_chunk(bam_id, offset, 2)
|
||||
assert len(double_chunk["ck_data"].split("\n")) == 2
|
||||
|
||||
def _display_chunk(self, dataset_id: str, offset: int, ck_size: int):
|
||||
return self.dataset_populator.display_chunk(dataset_id, offset, ck_size)
|
||||
|
||||
def test_tabular_chunking_through_display_endpoint(self, history_id):
|
||||
contents = textwrap.dedent(
|
||||
"""\
|
||||
1 2 3 4
|
||||
A B C D
|
||||
10 20 30 40
|
||||
"""
|
||||
)
|
||||
# test first chunk
|
||||
hda1 = self.dataset_populator.new_dataset(history_id, content=contents, wait=True, file_type="tabular")
|
||||
dataset_id = hda1["id"]
|
||||
chunk_1 = self._display_chunk(dataset_id, 0, 1)
|
||||
self._assert_has_keys(chunk_1, "offset", "ck_data")
|
||||
|
||||
assert chunk_1["ck_data"] == "1 2 3 4"
|
||||
assert chunk_1["offset"] == 14
|
||||
|
||||
chunk_2 = self._display_chunk(dataset_id, 14, 1)
|
||||
assert chunk_2["ck_data"] == "A B C D"
|
||||
assert chunk_2["offset"] == 28
|
||||
|
||||
chunk_3 = self._display_chunk(dataset_id, 28, 1)
|
||||
assert chunk_3["ck_data"] == "10 20 30 40"
|
||||
|
||||
def test_connectivity_table_chunking_through_display_endpoint(self, history_id):
|
||||
ct_dataset = self.dataset_populator.new_dataset(
|
||||
history_id, content=open(self.test_data_resolver.get_filename("1.ct"), "rb"), file_type="ct", wait=True
|
||||
)
|
||||
dataset_id = ct_dataset["id"]
|
||||
chunk_1 = self._display_chunk(dataset_id, 0, 1)
|
||||
self._assert_has_keys(chunk_1, "offset", "ck_data")
|
||||
|
||||
assert chunk_1["ck_data"] == "363 tmRNA"
|
||||
assert chunk_1["offset"] == 10, chunk_1
|
||||
|
||||
chunk_2 = self._display_chunk(dataset_id, 10, 1)
|
||||
assert chunk_2["ck_data"] == "1 G 0 2 359 1"
|
||||
|
||||
def test_head(self, history_id):
|
||||
hda1 = self.dataset_populator.new_dataset(history_id, wait=True)
|
||||
display_response = self._head(f"histories/{history_id}/contents/{hda1['id']}/display", {"raw": "True"})
|
||||
|
||||
@@ -451,9 +451,7 @@ class ImportExportTests(BaseHistories):
|
||||
raise SkipTest("skipping test_import_metadata_regeneration for task based...")
|
||||
history_name = f"for_import_metadata_regeneration_{uuid4()}"
|
||||
history_id = self.dataset_populator.new_history(name=history_name)
|
||||
self.dataset_populator.new_dataset(
|
||||
history_id, content=open(self.test_data_resolver.get_filename("1.bam"), "rb"), file_type="bam", wait=True
|
||||
)
|
||||
self.dataset_populator.new_bam_dataset(history_id, self.test_data_resolver)
|
||||
imported_history_id = self._reimport_history(history_id, history_name)
|
||||
self._assert_history_length(imported_history_id, 1)
|
||||
self._check_imported_dataset(history_id=imported_history_id, hid=1)
|
||||
|
||||
@@ -417,6 +417,11 @@ class BaseDatasetPopulator(BasePopulator):
|
||||
self.wait_for_tool_run(history_id, run_response, assert_ok=kwds.get("assert_ok", True))
|
||||
return run_response
|
||||
|
||||
def new_bam_dataset(self, history_id: str, test_data_resolver):
|
||||
return self.new_dataset(
|
||||
history_id, content=open(test_data_resolver.get_filename("1.bam"), "rb"), file_type="bam", wait=True
|
||||
)
|
||||
|
||||
def fetch(
|
||||
self,
|
||||
payload: dict,
|
||||
@@ -926,6 +931,19 @@ class BaseDatasetPopulator(BasePopulator):
|
||||
else:
|
||||
return display_response.content
|
||||
|
||||
def display_chunk(self, dataset_id: str, offset: int = 0, ck_size: Optional[int] = None) -> Dict[str, Any]:
|
||||
# use the dataset display API endpoint with the offset parameter to enable chunking
|
||||
# of the target dataset for certain datatypes
|
||||
kwds = {
|
||||
"offset": offset,
|
||||
}
|
||||
if ck_size is not None:
|
||||
kwds["ck_size"] = ck_size
|
||||
display_response = self._get(f"datasets/{dataset_id}/display", kwds)
|
||||
api_asserts.assert_status_code_is(display_response, 200)
|
||||
print(display_response.content)
|
||||
return display_response.json()
|
||||
|
||||
def get_history_dataset_source_transform_actions(self, history_id: str, **kwd) -> Set[str]:
|
||||
details = self.get_history_dataset_details(history_id, **kwd)
|
||||
if "sources" not in details:
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import json
|
||||
|
||||
from pysam import ( # type: ignore[attr-defined]
|
||||
AlignmentFile,
|
||||
view,
|
||||
@@ -63,3 +65,31 @@ def test_set_meta_header_info():
|
||||
"SQ": [{"SN": "ref", "LN": 45}, {"SN": "ref2", "LN": 40}],
|
||||
}
|
||||
assert dataset.metadata.reference_names == ["ref", "ref2"]
|
||||
|
||||
|
||||
def test_get_chunk():
|
||||
with get_dataset("bam_from_sam.bam") as dataset:
|
||||
chunk = _get_chunk_response(dataset, 0, 1)
|
||||
offset = chunk["offset"]
|
||||
|
||||
chunk2 = _get_chunk_response(dataset, offset, 1)
|
||||
|
||||
offset2 = chunk2["offset"]
|
||||
chunk3 = _get_chunk_response(dataset, offset2, 1)
|
||||
offset3 = chunk3["offset"]
|
||||
|
||||
assert offset < offset2
|
||||
assert offset2 < offset3
|
||||
|
||||
double_chunk = _get_chunk_response(dataset, offset, 2)
|
||||
double_chunk["ck_data"].startswith(chunk2["ck_data"])
|
||||
double_chunk["ck_data"].endswith(chunk3["ck_data"])
|
||||
|
||||
double_chunk_offset = double_chunk["offset"]
|
||||
assert offset3 == double_chunk_offset
|
||||
|
||||
|
||||
def _get_chunk_response(dataset, offset, chunk_size):
|
||||
b = Bam()
|
||||
chunk = b.get_chunk(None, dataset, offset, chunk_size)
|
||||
return json.loads(chunk)
|
||||
|
||||
Reference in New Issue
Block a user