From 80b3eff81a2cb587085e860ebd2b83d38bfd0b78 Mon Sep 17 00:00:00 2001 From: Phil Budne Date: Sat, 26 Sep 2026 01:17:01 -0400 Subject: [PATCH] Prepare for new version release (6.0) * Updated CHANGELOG.md * Renamed _parse_rate_limit to _parse_api_params * Have api_params() method strip wrapper * Added ApiParams and UserProfile/QuotaDict typing info * Removed stories_by_source_week method and type * Update version, add Repository to pyproject file --- CHANGELOG.md | 10 ++++++++ mediacloud/api.py | 28 ++++++++------------- mediacloud/test/api_directory_test.py | 4 +-- mediacloud/types.py | 36 +++++++++++++++++++++------ pyproject.toml | 3 ++- 5 files changed, 53 insertions(+), 28 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 493af0b..f8e8cff 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,16 @@ Version History =============== +Version 6 +--------- + +### v6.0.0 +* use new server api-params endpoint to warn users of new library and to acquire actual user rate limit +* remove stores_by_source_week (should have been removed in 5.0.0) +* add pytest --fast option for quick fire testing by impatient admin developers +* add typing for user_profile +* add Repository to pyproject.toml + Version 5 --------- diff --git a/mediacloud/api.py b/mediacloud/api.py index ebe1cfb..1a70c3a 100644 --- a/mediacloud/api.py +++ b/mediacloud/api.py @@ -12,8 +12,8 @@ from mediacloud.types import (Collection, CountOverTimePoint, JSONObj, LanguageCount, OffsetPage, PaginationToken, Source, SourceCount, SourceIntervalAttention, - SourceWeekAttention, Story, StoryCount, - VersionInfo) + Story, StoryCount, VersionInfo, + ApiParams, UserProfile) logger = logging.getLogger(__name__) @@ -69,7 +69,7 @@ def _make_session(self) -> None: try: raw = self.api_params() - per_minute = self._parse_rate_limit(raw) + per_minute = self._parse_api_params(raw) except: per_minute = 2 # old default @@ -78,11 +78,10 @@ def _make_session(self) -> None: self._session.headers.update(self._headers) self._per_minute = per_minute - def _parse_rate_limit(self, raw: JSONObj) -> int: + def _parse_api_params(self, params: ApiParams) -> int: # :return: rate limit in requests per minute # tries not to crash, and to handle bad data gracefully - pp = raw.get('params', {}) - if VERSION[0] == 'v' and (apc := pp.get('api-python-client')): + if VERSION[0] == 'v' and (apc := params.get('api-python-client')): try: apci = [int(x) for x in apc.split('.')] vers = [int(x) for x in VERSION[1:].split('.')] @@ -96,14 +95,14 @@ def _parse_rate_limit(self, raw: JSONObj) -> int: # only support per minute: per hour rates would allow LARGE bursts # django-ratelimit allows 10/5m, but django-smart-ratelimit may not?? per_minute = self.RATE_LIMIT_PER_MINUTE - if (qr := pp.get('query-rate')) and isinstance(qr, str): + if (qr := params.get('query-rate')) and isinstance(qr, str): sr = qr.split('/') # split rate if len(sr) == 2 and sr[0].isdigit() and sr[1] == 'm': # use RATE_LIMIT_PER_MINUTE as upper bound per_minute = min(int(sr[0]), per_minute) return per_minute - def user_profile(self) -> JSONObj: + def user_profile(self) -> UserProfile: # :return: basic info about the current user, including their roles return self._query('auth/profile') @@ -158,9 +157,10 @@ def _query(self, endpoint: str, params: Optional[Dict] = None, method: str = 'GE return j - def api_params(self) -> JSONObj: + def api_params(self) -> ApiParams: # :return: api parameters from server - return self._query('search/api-params') + results = self._query('search/api-params') + return results['params'] class DirectoryApi(BaseApi): @@ -176,6 +176,7 @@ def collection(self, collection_id: int) -> Collection: def collection_list(self, platform: Optional[str] = None, name: Optional[str] = None, limit: Optional[int] = 0, offset: Optional[int] = 0, source_id: Optional[int] = None) -> OffsetPage: + params: Dict[Any, Any] = dict(limit=limit, offset=offset) if name: params['name'] = name @@ -269,13 +270,6 @@ def story_count_over_time(self, query: str, start_date: dt.date, end_date: dt.da d['date'] = dt.date.fromisoformat(d['date'][:10]) return results['count_over_time']['counts'] - def stories_by_source_week(self, query: str, start_date: dt.date, end_date: dt.date, - collection_ids: Optional[List[int]] = [], source_ids: Optional[List[int]] = [], - platform: Optional[str] = None) -> List[SourceWeekAttention]: - params = self._prep_default_params(query, start_date, end_date, collection_ids, source_ids, platform) - results = self._query('search/count-by-source-week', params) - return results['source-week-attention'] - def stories_by_source_over_interval(self, query: str, start_date: dt.date, end_date: dt.date, collection_ids: Optional[List[int]] = [], source_ids: Optional[List[int]] = [], platform: Optional[str] = None, interval: Optional[str] = None) -> List[SourceIntervalAttention]: diff --git a/mediacloud/test/api_directory_test.py b/mediacloud/test/api_directory_test.py index bbeb9fc..0e75dfc 100644 --- a/mediacloud/test/api_directory_test.py +++ b/mediacloud/test/api_directory_test.py @@ -5,7 +5,7 @@ from unittest import TestCase import mediacloud.api -from mediacloud.types import JSONObj +from mediacloud.types import ApiParams TEST_COLLECTION_ID = 34412234 # US -National sources TEST_SOURCE_ID = 1095 # cnn.com @@ -15,7 +15,7 @@ class MyDirectoryApi(mediacloud.api.DirectoryApi): BASE_API_URL = os.getenv("MC_API_BASE_URL", "https://search.mediacloud.org/api/") RATE_LIMIT_PER_MINUTE = 60 # upper bound - def _parse_rate_limit(self, raw: JSONObj) -> int: + def _parse_api_params(self, raw: ApiParams) -> int: # ignore data from api-params call # (rates only enforced for searches anyway!) return self.RATE_LIMIT_PER_MINUTE diff --git a/mediacloud/types.py b/mediacloud/types.py index 802d796..9a4227b 100644 --- a/mediacloud/types.py +++ b/mediacloud/types.py @@ -41,14 +41,6 @@ class LanguageCount(TypedDict, total=False): value: int -class SourceWeekAttention(TypedDict, total=False): - media_name: str - week: str - matching_stories: int - total_stories: int - ratio: float - - class SourceIntervalAttention(TypedDict, total=False): media_name: str interval: str @@ -100,3 +92,31 @@ class VersionInfo(TypedDict, total=False): GIT_REV: str now: float version: str + + +class QuotaDict(TypedDict, total=False): + provider: str + hits: int # usage this week + week: str # YYYY-MM-DD + limit: int # weekly quota + + +class UserProfile(TypedDict, total=False): + id: int + username: str + is_staff: bool + is_superuser: bool + groups: list[str] + quota: QuotaDict + + +ApiParams = TypedDict( + 'ApiParams', + { + # suggested Python client version + "api-python-client": "str", + # user's actual query rate limit (digits/m) + "query-rate": "str", + }, + total=False +) diff --git a/pyproject.toml b/pyproject.toml index 79ccf07..79de114 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -7,7 +7,7 @@ include = ["mediacloud/py.typed"] [project] name = "mediacloud" -version = "5.1.0" +version = "6.0.0" authors = [ {name = "Rahul Bhargava", email = "r.bhargava@northeastern.edu"}, ] @@ -37,6 +37,7 @@ test = [ [project.urls] "Homepage" = "https://mediacloud.org" +"Repository" = "https://github.com/mediacloud/api-client" "Bug Tracker" = "https://github.com/mediacloud/api-client/issues" [tool.mypy]