Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 5 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -261,7 +261,11 @@ granules = api.get(25000)
granules = api.get_all() # this is a shortcut for api.get(api.hits())
```

By default the responses will return as json and be accessible as a list of python dictionaries. Other formats can be
By default the responses will return as json and be accessible as a list of python dictionaries.
The `umm_json` and versioned `umm_json_vX_Y` formats also return individual dictionaries,
each containing the record's `meta` and `umm` fields. This applies to both `results()`
and `get()`; callers no longer need to parse UMM JSON pages with `json.loads`.
Other formats can be
specified before making the request:

```python
Expand Down
19 changes: 13 additions & 6 deletions cmr/queries.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,11 +90,12 @@ def get(self, limit: int = 2000) -> Sequence[Any]:
# list is the number of *pages* fetched, not the number of *items*.
n_results += page_size

results.extend(
response.json()["feed"]["entry"]
if self._format == "json"
else [response.text]
)
if self._format == "json":
results.extend(response.json()["feed"]["entry"])
elif self._format.startswith("umm_json"):
results.extend(response.json()["items"])
else:
results.append(response.text)

if cmr_search_after := response.headers.get("cmr-search-after"):
headers["cmr-search-after"] = cmr_search_after
Expand Down Expand Up @@ -153,6 +154,10 @@ def results(self, page_size: int = 2000) -> Iterator[Any]:
In this case, the iterator may produce as many elements as there
are results matching the query criteria.

For `"umm_json"` and versioned `"umm_json_v*"` formats, each element
is an entry from the response's "items" array, including both
its "meta" and "umm" fields.

For all other formats, each element produced by the returned
iterator is an unparsed (text) page of results (i.e., the caller
is responsible for parsing the page of results into individual
Expand All @@ -175,6 +180,8 @@ def results(self, page_size: int = 2000) -> Iterator[Any]:

if self._format == "json":
yield from response.json()["feed"]["entry"]
elif self._format.startswith("umm_json"):
yield from response.json()["items"]
else:
yield response.text

Expand Down Expand Up @@ -1190,7 +1197,7 @@ def get(self, limit: int = 2000) -> Sequence[Any]:
)
response.raise_for_status()

if self._format == "json":
if self._format == "json" or self._format.startswith("umm_json"):
latest = response.json()['items']
else:
latest = [response.text]
Expand Down
6 changes: 2 additions & 4 deletions tests/test_granule.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,10 +90,8 @@ def test_revision_date(self):
"umm_json").get_all()
granule_dict = {}
for granule in granules:
granule_json = json.loads(granule)
for item in granule_json["items"]:
native_id = item["meta"]["native-id"]
granule_dict[native_id] = item
native_id = granule["meta"]["native-id"]
granule_dict[native_id] = granule

self.assertIn("SWOT_L2_HR_RiverSP_Reach_017_312_AS_20240630T042656_20240630T042706_PIC0_01_swot",
granule_dict.keys())
Expand Down
15 changes: 4 additions & 11 deletions tests/test_multiple_queries.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,4 @@
import inspect
import json
import os
from datetime import datetime
from itertools import islice
Expand Down Expand Up @@ -39,10 +38,7 @@ def test_get_unparsed_format(self):
"""
api = GranuleQuery().format("umm_json")

pages = api.short_name("MOD02QKM").get(2)
granules = [
granule for page in pages for granule in json.loads(page)["items"]
]
granules = api.short_name("MOD02QKM").get(2)
self.assertEqual(2, len(granules))
assert_unique_granules_from_results(granules)
# Assert that we performed only 1 search request
Expand All @@ -51,17 +47,14 @@ def test_get_unparsed_format(self):

def test_results_unparsed_format(self):
"""
If we execute a get for an unparsed format we expect pages to be returned instead of items
UMM JSON results yield individual items across all recorded pages.
"""
api = GranuleQuery().format("umm_json")

pages = list(islice(api.short_name("MOD02QKM").results(page_size=2), 10))
granules = [
granule for page in pages for granule in json.loads(page)["items"]
]
granules = list(islice(api.short_name("MOD02QKM").results(page_size=2), 20))
self.assertEqual(20, len(granules))
assert_unique_granules_from_results(granules)
# Assert that we performed 5 search requests
# Assert that we performed 10 search requests
self.assertEqual(10, len(self.cassette))
self.assertIsNone((api.headers or {}).get("cmr-search-after"))

Expand Down
34 changes: 34 additions & 0 deletions tests/test_umm_results.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
from unittest.mock import Mock

import pytest

from cmr.queries import CollectionQuery, ServiceQuery, ToolQuery, VariableQuery


@pytest.mark.parametrize("query_type", [CollectionQuery, ServiceQuery, ToolQuery, VariableQuery])
@pytest.mark.parametrize("output_format", ["umm_json", "umm_json_v1_9"])
@pytest.mark.parametrize("method", ["get", "results"])
def test_umm_results_are_individual_items(monkeypatch, query_type, output_format, method):
items = [{"meta": {"concept-id": "C1"}, "umm": {"ShortName": "first"}},
{"meta": {"concept-id": "C2"}, "umm": {"ShortName": "second"}}]
pages = [Mock(headers={"cmr-search-after": "next"}, text="first page"),
Mock(headers={}, text="second page"), Mock(headers={}, text="empty page")]
for page, records in zip(pages, [[items[0]], [items[1]], []]):
page.json.return_value = {"items": records}
request = Mock(side_effect=pages)
monkeypatch.setattr("cmr.queries.requests.get", request)
query = query_type().format(output_format)

if method == "results":
actual = list(query.results(page_size=1))
else:
actual = query.get(limit=2001 if query_type is CollectionQuery else 2)

assert actual == items
assert request.call_count == 2


def test_xml_results_remain_unparsed(monkeypatch):
response = Mock(headers={}, text="<results/>")
monkeypatch.setattr("cmr.queries.requests.get", Mock(return_value=response))
assert list(CollectionQuery().format("xml").results()) == ["<results/>"]