Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -100,3 +100,4 @@ _version.py
tests/hdx/api/gsheet_auth.json

.vscode/settings.json
.claude/
43 changes: 43 additions & 0 deletions documentation/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@ upload your datasets to HDX.
- [Showcase Management](#showcase-management)
- [User Management](#user-management)
- [Organization Management](#organization-management)
- [Location Management](#location-management)
- [Vocabulary Management](#vocabulary-management)
- [Pipeline State](#pipeline-state)
- [Working Examples](#working-examples)
Expand Down Expand Up @@ -1063,6 +1064,48 @@ You can delete a user from an organization:

organization.delete_user("USER ID")

## Location Management

The **Location** class wraps HDX locations (CKAN groups eg. countries, "world"). Reading
locations does not require any special permissions, so these are the operations you will
typically use. Creating, updating and deleting locations is restricted to HDX sysadmins
and works the same way as for other HDX objects (see
[Operations on HDX Objects](#operations-on-hdx-objects)).

You can read a single location:

location = Location.read_from_hdx("LOCATION_ID_OR_NAME")

You can get all location names in HDX:

location_names = Location.get_all_location_names(**kwargs)

Passing `all_fields=True` (plus optionally `include_extras=True` etc.) returns the full
location dictionaries instead of just names. Various additional arguments (`**kwargs`)
can be supplied and are detailed in the API documentation.

You can fetch the countries that are currently active on HDX's Data Grid (a live lookup,
since Data Grid membership changes over time, rather than a fixed list):

country_codes = Location.get_data_grid_countries()

This returns a sorted list of 3-letter location codes.

You can get the datasets belonging to a location as follows:

datasets = location.get_datasets(**kwargs)

Various additional arguments (`**kwargs`) can be supplied. These are detailed in the API
documentation.

You can autocomplete a location name:

matches = Location.autocomplete("NAME")

For simple name/code lookups against a cached list of valid locations (eg. converting
between a country name and its HDX code) rather than the full **Location** object, see
the **Locations** helper class instead.

## Vocabulary Management

The **Vocabulary** class enables you to manage CKAN vocabularies, creating, deleting and
Expand Down
4 changes: 4 additions & 0 deletions src/hdx/api/hdx_base_configuration.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,10 @@ organization:
- name
- title
- description
group:
required_fields:
- name
- title
"resource view":
required_fields:
- resource_id
Expand Down
7 changes: 3 additions & 4 deletions src/hdx/api/locations.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from collections.abc import Sequence

from hdx.api.configuration import Configuration
from hdx.data.location import Location


class Locations:
Expand All @@ -22,10 +23,8 @@ def validlocations(cls, configuration=None) -> list[dict]:
A list of valid locations
"""
if cls._validlocations is None:
if configuration is None:
configuration = Configuration.read()
cls._validlocations = configuration.call_remoteckan(
"group_list", {"all_fields": True}
cls._validlocations = Location.get_all_location_names(
configuration=configuration, all_fields=True
)
return cls._validlocations

Expand Down
237 changes: 237 additions & 0 deletions src/hdx/data/location.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,237 @@
"""Location class containing all logic for creating, checking, and updating locations."""

import logging
from collections.abc import Sequence
from typing import Any, Optional

from hdx.api.configuration import Configuration
from hdx.data.hdxobject import HDXObject

logger = logging.getLogger(__name__)


class Location(HDXObject):
"""Location class containing all logic for creating, checking, and updating locations.
A location is a CKAN group.

Args:
initial_data: Initial location metadata dictionary. Defaults to None.
configuration: HDX configuration. Defaults to global configuration.
"""

def __init__(
self,
initial_data: dict | None = None,
configuration: Configuration | None = None,
) -> None:
if not initial_data:
initial_data = {}
super().__init__(initial_data, configuration=configuration)

@staticmethod
def actions() -> dict[str, str]:
"""Dictionary of actions that can be performed on object

Returns:
Dictionary of actions that can be performed on object
"""
return {
"show": "group_show",
"update": "group_update",
"create": "group_create",
"delete": "group_delete",
"list": "group_list",
"autocomplete": "group_autocomplete",
}

@classmethod
def read_from_hdx(
cls, identifier: str, configuration: Configuration | None = None
) -> Optional["Location"]:
"""Reads the location given by identifier from HDX and returns Location object

Args:
identifier: Identifier of location
configuration: HDX configuration. Defaults to global configuration.

Returns:
Location object if successful read, None if not
"""
return cls._read_from_hdx_class("group", identifier, configuration)

def check_required_fields(self, ignore_fields: Sequence[str] = ()) -> None:
"""Check that metadata for location is complete. The parameter ignore_fields should
be set if required to any fields that should be ignored for the particular operation.

Args:
ignore_fields: Fields to ignore. Default is ().

Returns:
None
"""
self._check_required_fields("group", ignore_fields)

def update_in_hdx(self, **kwargs: Any) -> None:
"""Check if location exists in HDX and if so, update location

Returns:
None
"""
self._update_in_hdx("group", "id", **kwargs)

def create_in_hdx(self, **kwargs: Any) -> None:
"""Check if location exists in HDX and if so, update it, otherwise create location

Returns:
None
"""
self._create_in_hdx("group", "id", "name", **kwargs)

def delete_from_hdx(self) -> None:
"""Deletes a location from HDX.

Returns:
None
"""
self._delete_from_hdx("group", "id")

def get_datasets(self, query: str = "*:*", **kwargs: Any) -> list["Dataset"]: # noqa: F821
"""Get list of datasets in location

Args:
query: Restrict datasets returned to this query (in Solr format). Defaults to '*:*'.
**kwargs: See below
sort (string): Sorting of the search results. Defaults to 'relevance asc, metadata_modified desc'.
rows (int): Number of matching rows to return. Defaults to all datasets (sys.maxsize).
start (int): Offset in the complete result for where the set of returned datasets should begin
facet (string): Whether to enable faceted results. Default to True.
facet.mincount (int): Minimum counts for facet fields should be included in the results
facet.limit (int): Maximum number of values the facet fields return (- = unlimited). Defaults to 50.
facet.field (list[str]): Fields to facet upon. Default is empty.
use_default_schema (bool): Use default package schema instead of custom schema. Defaults to False.

Returns:
List of datasets in location
"""
import hdx.data.dataset # avoid circular import

return hdx.data.dataset.Dataset.search_in_hdx(
query=query,
configuration=self.configuration,
fq=f"groups:{self.data['name']}",
**kwargs,
)

@staticmethod
def get_all_location_names(
configuration: Configuration | None = None, **kwargs: Any
) -> list[str]:
"""Get all location names in HDX

Args:
configuration: HDX configuration. Defaults to global configuration.
**kwargs: See below
sort (str): Sort the search results according to field name and sort-order. Allowed fields are ‘name’, ‘package_count’ and ‘title’. Defaults to 'name asc'.
groups (list[str]): List of names of the groups to return.
all_fields (bool): Return group dictionaries instead of just names. Only core fields are returned - get some more using the include_* options. Defaults to False.
include_extras (bool): If all_fields, include the group extra fields. Defaults to False.
include_tags (bool): If all_fields, include the group tags. Defaults to False.
include_groups (bool): If all_fields, include the groups the groups are in. Defaults to False.

Returns:
List of all location names in HDX
"""
location = Location(configuration=configuration)

# Early return for the standard, non-paginated case
if not kwargs.get("all_fields"):
return location._write_to_hdx("list", kwargs)

compiled_locations = {}
# Extract limit and offset to dictate our paging sizes, defaulting to 400 and 0
page_size = kwargs.pop("limit", 400)
current_offset = kwargs.pop("offset", 0)

# 1. Fetch in pages using the provided/default page size
while True:
page_kwargs = kwargs.copy()
page_kwargs["limit"] = page_size
page_kwargs["offset"] = current_offset

page_data = location._write_to_hdx("list", page_kwargs)

if not page_data:
break

# Store by name to automatically deduplicate offset shifts
for loc in page_data:
compiled_locations[loc["name"]] = loc

if len(page_data) < page_size:
break

current_offset += page_size

# 2. Establish Ground Truth LAST (without limit/offset constraints)
truth_kwargs = kwargs.copy()
truth_kwargs["all_fields"] = False

ground_truth_names = location._write_to_hdx("list", truth_kwargs)

# 3 & 4. Check for missing IDs, fetch individually, and implicitly prune deletes
final_locations = []
for name in ground_truth_names:
if name in compiled_locations:
final_locations.append(compiled_locations[name])
else:
try:
missing_location = location._write_to_hdx("show", {"id": name})
if missing_location:
final_locations.append(missing_location)
except Exception:
pass

return final_locations

@staticmethod
def get_data_grid_countries(
configuration: Configuration | None = None,
) -> list[str]:
"""Fetch HDX's active Data Grid countries (3-letter group names).

Treated as a live/short-TTL lookup rather than a frozen static list, since Data Grid
membership changes over time.

Args:
configuration: HDX configuration. Defaults to global configuration.

Returns:
Sorted list of active Data Grid country (group) names
"""
groups = Location.get_all_location_names(
configuration=configuration, all_fields=True, include_extras=True
)
return sorted(
group["name"]
for group in groups
if len(group["name"]) == 3 and group.get("data_completeness") == "active"
)

@classmethod
def autocomplete(
cls,
name: str,
limit: int = 20,
configuration: Configuration | None = None,
) -> list:
"""Autocomplete a location name and return matches

Args:
name: Name to autocomplete
limit: Maximum number of matches to return
configuration: HDX configuration. Defaults to global configuration.

Returns:
Autocomplete matches
"""
return cls._autocomplete(name, limit, configuration)
31 changes: 31 additions & 0 deletions tests/fixtures/location/location_list_all_fields.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
- id: afg
name: afg
title: Afghanistan
display_name: Afghanistan
type: group
is_organization: false
state: active
data_completeness: active
description: ""
package_count: 480

- id: zmb
name: zmb
title: Zambia
display_name: Zambia
type: group
is_organization: false
state: active
data_completeness: inactive
description: ""
package_count: 120

- id: world
name: world
title: World
display_name: World
type: group
is_organization: false
state: active
description: ""
package_count: 12
11 changes: 11 additions & 0 deletions tests/fixtures/location/location_show_results.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
id: "f8b0e58f-9be5-4b25-9d5f-1c8b6d8b6a0a"
name: gnq
title: "Equatorial Guinea"
display_name: "Equatorial Guinea"
type: group
is_organization: false
state: active
data_completeness: active
description: ""
package_count: 5
created: "2015-01-09T14:44:54.006612"
5 changes: 5 additions & 0 deletions tests/hdx/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -119,6 +119,11 @@ def json(self):
"description": "We do humanitarian work",
}

location_data = {
"name": "MyLocation1",
"title": "My Location",
}

user_data = {
"name": "MyUser1",
"email": "xxx@yyy.com",
Expand Down
Loading