diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 34558c4f..5e00ec6e 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -67,7 +67,6 @@ The ROR API provides: | App server | Phusion Passenger + Nginx (`vendor/docker/webapp.conf`) | | Container | Docker (`Dockerfile` based on `phusion/passenger-python312:3.2.0`) | | Observability | Sentry (`sentry-sdk` 1.45.1), django-prometheus 2.4.1 | -| Feature flags | LaunchDarkly (`rorapi/common/features.py`; `launchdarkly-server-sdk` 7.6.1) | | Email | django-ses 4.8.0 (client ID registration emails) | | External packages | `update_address` (Geonames enrichment), `jsonschema` 3.2.0, `rapidfuzz` 3.6.1, `boto3` (unpinned), `pandas` 2.2.3 | @@ -107,7 +106,7 @@ ror-api/ │ │ ├── record_template.json │ │ ├── ror_schema_v2_1.json # Vendored JSON schema for write validation │ │ └── index_template_es7.json # ES index template + mappings -│ ├── management/commands/ # CLI indexing and legacy GRID tools +│ ├── management/commands/ # CLI indexing and data setup │ ├── migrations/ # Django migrations (Client model) │ └── tests/ # Unit, integration, functional, affiliation suites └── vendor/docker/ # Nginx, env, Terraform var templates for deploy @@ -333,7 +332,6 @@ Loaded from environment and optional root `.env` file (`python-dotenv`). | `ROUTE_USER`, `TOKEN` | Admin API authentication | | `ROR_BASE_URL` | Base URL configuration | | `SENTRY_DSN` | Error reporting | -| `LAUNCH_DARKLY_KEY` | Feature flags | | `SINGLE_SEARCH_DEFAULT` | Default affiliation matcher (`True`/`False`) | | `ENABLE_BEHAVIORAL_LIMITING` | Rate limiting toggle (edge behavior) | | `SECRET_KEY` | Django secret (falls back to a hardcoded default if unset; `DEBUG` is always `False`) | @@ -400,9 +398,7 @@ Deploy mechanism: GitHub Action updates `_ror-api-*.auto.tfvars` in the `new-dep ## Legacy Code -Commands prefixed with `legacy*` (GRID conversion, old upgrade paths) are **non-functional** — referenced data was moved to ror-data. GRID-based generation ended March 2022. Do not extend or rely on these unless explicitly reviving historical tooling. - -`settings.py` still contains commented GRID/ROR_DUMP version history for reference. `GRID_REMOVED_IDS` is an empty list retained for a check in `retrieve_organization`. +GRID-based generation ended March 2022. The `legacy*` management commands, the LaunchDarkly call site, and the empty `GRID_REMOVED_IDS` check have been removed. `settings.py` still contains a short comment noting that ROR is no longer based on GRID. Historical GRID/ROR dump files live in [ror-data](https://github.com/ror-community/ror-data). --- diff --git a/README.md b/README.md index b517d280..a29e935a 100644 --- a/README.md +++ b/README.md @@ -107,49 +107,9 @@ The API uses the v2 schema only. Use `-s 2` when indexing a data dump. A v2 form python manage.py setup v1.32-2023-09-14-ror-data -s 2 -t -## LEGACY: Converting GRID data to ROR (process used prior to Mar 2022) +## GRID history (prior to Mar 2022) -Steps used prior to Mar 2022: -- Convert latest GRID dataset to ROR (including assigning ROR IDs) -- Generate ROR data dump -- Index ROR data dump into Elastic Search - -As of Mar 2022 ROR is no longer based on GRID. Record additions/updates and data deployment is now managed in https://github.com/ror-community/ror-records using the ```indexror``` command described above. - -Steps below no longer work, as data files have been moved to [ror-data](https://github.com/ror-community/ror-data). This information is being maintained for historical purposes. - -Management commands used in this process no longer work and are pre-pended with "legacy". - - -To import GRID data, you need a system where `setup` has been run successfully. Then first update the `GRID` variable in `settings.py`, e.g. - -``` -GRID = { - 'VERSION': '2020-03-15', - 'URL': 'https://digitalscience.figshare.com/ndownloader/files/22091379' -} -``` - -And, also in `settings.py`, set the `ROR_DUMP` variable, e.g. - -``` -ROR_DUMP = {'VERSION': '2020-04-02'} -``` - -Then run this command: `./manage.py upgrade`. - -You should see this in the console: - -``` -Downloading GRID version 2020-03-15 -Converting GRID dataset to ROR schema -ROR dataset created -ROR dataset ZIP archive created -``` - -This will create a new `data/ror-2020-03-15` folder, containing a `ror.json` and `ror.zip`. To finish the process, add the new folder to git and push to the GitHub repo. - -To install the updated ROR data, run `./manage.py setup`. +Before March 2022, ROR records were derived from GRID: convert the GRID dataset, generate a dump, and index it. That pipeline and its management commands have been removed. Record additions/updates and data deployment are now managed in https://github.com/ror-community/ror-records using the `indexror` command described above. Historical GRID/ROR dump files live in [ror-data](https://github.com/ror-community/ror-data). ## Create new record file (v2 only) diff --git a/rorapi/common/features.py b/rorapi/common/features.py deleted file mode 100644 index 680402f5..00000000 --- a/rorapi/common/features.py +++ /dev/null @@ -1,6 +0,0 @@ -import ldclient -from ldclient.config import Config -from rorapi.settings import LAUNCH_DARKLY_KEY - -ldclient.set_config(Config(LAUNCH_DARKLY_KEY)) -launch_darkly_client = ldclient.get() \ No newline at end of file diff --git a/rorapi/common/queries.py b/rorapi/common/queries.py index bd4d02aa..1577c0e0 100644 --- a/rorapi/common/queries.py +++ b/rorapi/common/queries.py @@ -1,5 +1,6 @@ import re import json +import unicodedata from titlecase import titlecase from collections import defaultdict @@ -11,7 +12,7 @@ Organization as OrganizationV2, ListResult as ListResultV2 ) -from rorapi.settings import GRID_REMOVED_IDS, ROR_API, ES_VARS +from rorapi.settings import ROR_API, ES_VARS from rorapi.common.es_utils import ESQueryBuilder from urllib.parse import unquote @@ -187,6 +188,14 @@ def validate(params): return Errors(errors) if errors else None +def nfc(value): + """Normalizes a string to Unicode NFC so decomposed (NFD) input matches + the precomposed characters indexed in Elasticsearch""" + if isinstance(value, str): + return unicodedata.normalize("NFC", value) + return value + + def build_search_query(params): """Builds search query from API parameters""" @@ -198,20 +207,21 @@ def build_search_query(params): del params["all_status"] if "query.advanced" in params: - qb.add_string_query_advanced(params.get("query.advanced")) + qb.add_string_query_advanced(nfc(params.get("query.advanced"))) elif "query" in params: - ror_id = get_ror_id(params.get("query")) + query = nfc(params.get("query")) + ror_id = get_ror_id(query) if ror_id is not None: qb.add_id_query(ror_id) else: - qb.add_string_query(params.get("query")) + qb.add_string_query(query) else: qb.add_match_all_query() if "filter" in params or (not "all_status" in params): filters = [ f.split(":") - for f in filter_string_to_list(params.get("filter", "")) + for f in filter_string_to_list(nfc(params.get("filter", ""))) if f ] # normalize filter values based on casing conventions used in ROR records @@ -247,10 +257,10 @@ def build_search_query(params): qb.add_aggregations( [ - ("types", "types"), - ("countries", "locations.geonames_details.country_code"), - ("continents", "locations.geonames_details.continent_code"), - ("statuses", "status"), + ("types", "types.raw"), + ("countries", "locations.geonames_details.country_code.raw"), + ("continents", "locations.geonames_details.continent_code.raw"), + ("statuses", "status.raw"), ] ) @@ -280,19 +290,6 @@ def search_organizations(params): def retrieve_organization(ror_id): """Retrieves the organization of the given ROR ID""" - if any(ror_id in ror_id_url for ror_id_url in GRID_REMOVED_IDS): - return ( - Errors( - [ - "ROR ID '{}' was removed by GRID during the time period (Jan 2019-Mar 2022) " - "that ROR was synced with GRID. We are currently working with the ROR Curation Advisory Board " - "to restore these records and expect to complete this work in 2022".format( - ror_id - ) - ] - ), - None, - ) search = build_retrieve_query(ror_id) results = search.execute() total = results.hits.total.value diff --git a/rorapi/common/views.py b/rorapi/common/views.py index 439a348a..68aa99d0 100644 --- a/rorapi/common/views.py +++ b/rorapi/common/views.py @@ -5,6 +5,7 @@ from django.views import View from django.shortcuts import redirect from rest_framework.permissions import BasePermission +from rest_framework.throttling import AnonRateThrottle from rest_framework.views import APIView from rest_framework.parsers import FormParser, MultiPartParser from rorapi.settings import DATA @@ -43,7 +44,16 @@ from rorapi.v2.models import Client from rorapi.v2.serializers import ClientSerializer + +class ClientRegistrationThrottle(AnonRateThrottle): + """Tight anonymous limit for client-ID registration (settings.py left unchanged).""" + + rate = "5/hour" + + class ClientRegistrationView(APIView): + throttle_classes = [ClientRegistrationThrottle] + def post(self, request, version='v2'): serializer = ClientSerializer(data=request.data) if serializer.is_valid(): diff --git a/rorapi/management/commands/legacyconvertgrid.py b/rorapi/management/commands/legacyconvertgrid.py deleted file mode 100644 index cbddbff4..00000000 --- a/rorapi/management/commands/legacyconvertgrid.py +++ /dev/null @@ -1,209 +0,0 @@ -import base32_crockford -import json -import os.path -import random -import zipfile -import re -from rorapi.settings import ES, ES_VARS, ROR_API, GRID, ROR_DUMP - -from django.core.management.base import BaseCommand - -# Previously used to convert latest GRID dataset configured in settings.py -# to ROR and assign ROR IDs to each GRID org -# As of Mar 2022 ROR is no longer based on GRID -# New records are now created in https://github.com/ror-community/ror-records and pushed to S3 -# Individual record files in S3 are indexed with indexror.py -# Entire dataset zip files in https://github.com/ror-community/ror-data -# can be indexed with setup.py, which uses indexrordump.py - -def generate_ror_id(): - """Generates random ROR ID. - - The checksum calculation is copied from - https://github.com/datacite/base32-url/blob/master/lib/base32/url.rb - to maintain the compatibility with previously generated ROR IDs. - """ - - n = random.randint(0, 200000000) - n_encoded = base32_crockford.encode(n).lower().zfill(6) - checksum = str(98 - ((n * 100) % 97)).zfill(2) - return '{}0{}{}'.format(ROR_API['ID_PREFIX'], n_encoded, checksum) - - -def get_ror_id(grid_id, es): - """Maps GRID ID to ROR ID. - - If given GRID ID was indexed previously, corresponding ROR ID is obtained - from the index. Otherwise, new ROR ID is generated. - """ - - s = ES.search(ES_VARS['INDEX'], - body={'query': { - 'term': { - 'external_ids.GRID.all': grid_id - } - }}) - if s['hits']['total'] == 1: - return s['hits']['hits'][0]['_id'] - return generate_ror_id() - - -def geonames_city(geonames_city): - geonames = ["geonames_admin1", "geonames_admin2"] - geonames_attributes = ["id", "name", "ascii_name", "code"] - nuts = ["nuts_level1", "nuts_level2", "nuts_level3"] - nuts_attributes = ["code", "name"] - geonames_city_hsh = {} - for k, v in geonames_city.items(): - if (k in geonames): - if isinstance(v, dict): - geonames_city_hsh[k] = { - i: v.get(i, None) - for i in geonames_attributes - } - elif v is None: - geonames_city_hsh[k] = {i: None for i in geonames_attributes} - elif (k in nuts): - if isinstance(v, dict): - geonames_city_hsh[k] = { - i: v.get(i, None) - for i in nuts_attributes - } - elif v is None: - geonames_city_hsh[k] = {i: None for i in nuts_attributes} - else: - geonames_city_hsh[k] = v - return geonames_city_hsh - - -def addresses(location): - line = "" - address = ["line_1", "line_2", "line_3"] - combine_lines = address + ["country", "country_code"] - geonames_admin = ["id", "code", "name", "ascii_name"] - nuts = ["code", "name"] - new_addresses = [] - hsh = {} - hsh["line"] = None - for h in location: - for k, v in h.items(): - if not (k in combine_lines) and (k != "geonames_city"): - v = v if v != "" else None - hsh[k] = v - elif k == "geonames_city": - if isinstance(v, dict): - hsh[k] = geonames_city(v) - elif v is None: - hsh[k] = {} - elif (k in combine_lines): - n = [] - for i in address: - if not (h[i] is None): - n.append(h[i]) - line = " ".join(n) - line = re.sub(' +', ' ', line) - if (len(line) == 1 and line == " "): - line = line.strip() - line = line if len(line) > 0 else None - hsh["line"] = line - new_addresses.append(hsh) - return new_addresses - - -def convert_organization(grid_org, es): - """Converts the organization metadata from GRID schema to ROR schema.""" - return { - 'id': - get_ror_id(grid_org['id'], ES), - 'name': - grid_org['name'], - 'types': - grid_org['types'], - 'links': - grid_org['links'], - 'aliases': - grid_org['aliases'], - 'acronyms': - grid_org['acronyms'], - 'status': - grid_org['status'], - 'wikipedia_url': - grid_org['wikipedia_url'], - 'labels': - grid_org['labels'], - 'email_address': - grid_org['email_address'], - 'ip_addresses': - grid_org['ip_addresses'], - 'established': - grid_org['established'], - 'country': { - 'country_code': grid_org['addresses'][0]['country_code'], - 'country_name': grid_org['addresses'][0]['country'] - }, - 'relationships': - grid_org["relationships"], - 'addresses': - addresses(grid_org["addresses"]), - 'external_ids': - getExternalIds( - dict(grid_org.get('external_ids', {}), - GRID={ - 'preferred': grid_org['id'], - 'all': grid_org['id'] - })) - } - - -def getExternalIds(external_ids): - if 'ROR' in external_ids: del external_ids['ROR'] - return external_ids - - -def get_ids(data): - ids = {} - for d in data: - ids[d['external_ids']['GRID']['all']] = d['id'] - return ids - - -def get_grid(record, ids): - if record['relationships']: - for r in record['relationships']: - r['id'] = ids[r['id']] - - return record - - -class Command(BaseCommand): - help = 'Converts GRID dataset to ROR schema' - - def handle(self, *args, **options): - os.makedirs(ROR_DUMP['DIR'], exist_ok=True) - # make sure we are not overwriting an existing ROR JSON file - # with new ROR identifiers - if zipfile.is_zipfile(ROR_DUMP['ROR_ZIP_PATH']): - self.stdout.write('ROR dataset already exists') - return - - if not os.path.isfile(ROR_DUMP['ROR_JSON_PATH']): - with open(GRID['GRID_JSON_PATH'], 'r') as it: - grid_data = json.load(it) - - self.stdout.write('Converting GRID dataset to ROR schema') - intermediate_ror_data = [ - convert_organization(org, ES) - for org in grid_data['institutes'] if org['status'] == 'active' - ] - ids = get_ids(intermediate_ror_data) - ror_data = [get_grid(rec, ids) for rec in intermediate_ror_data] - with open(ROR_DUMP['ROR_JSON_PATH'], 'w') as outfile: - json.dump(ror_data, outfile, indent=4) - self.stdout.write('ROR dataset created') - - # generate zip archive - with zipfile.ZipFile(ROR_DUMP['ROR_ZIP_PATH'], 'w') as zipArchive: - zipArchive.write(ROR_DUMP['ROR_JSON_PATH'], - arcname='ror.json', - compress_type=zipfile.ZIP_DEFLATED) - self.stdout.write('ROR dataset ZIP archive created') diff --git a/rorapi/management/commands/legacydownloadgrid.py b/rorapi/management/commands/legacydownloadgrid.py deleted file mode 100644 index 0c75bdea..00000000 --- a/rorapi/management/commands/legacydownloadgrid.py +++ /dev/null @@ -1,36 +0,0 @@ -import os -import requests -import zipfile - -from django.core.management.base import BaseCommand -from rorapi.settings import GRID - -# Previously used to download latest GRID dataset configured in settings.py -# which was used to generate a new ROR datasets -# As of Mar 2022 ROR is no longer based on GRID -# New records are now created in https://github.com/ror-community/ror-records and pushed to S3 -# Individual record files in S3 are indexed with indexror.py -# Entire dataset zip files in https://github.com/ror-community/ror-data -# can be indexed with setup.py, which uses indexrordump.py - -class Command(BaseCommand): - help = 'Downloads GRID dataset' - - def handle(self, *args, **options): - os.makedirs(GRID['DIR'], exist_ok=True) - - # make sure we are not overwriting an existing ROR JSON file - # with new ROR identifiers - if zipfile.is_zipfile(GRID['GRID_ZIP_PATH']): - self.stdout.write('Already downloaded GRID version {}'.format( - GRID['VERSION'])) - return - - self.stdout.write('Downloading GRID version {}'.format( - GRID['VERSION'])) - r = requests.get(GRID['URL']) - with open(GRID['GRID_ZIP_PATH'], 'wb') as f: - f.write(r.content) - - with zipfile.ZipFile(GRID['GRID_ZIP_PATH'], 'r') as zip_ref: - zip_ref.extractall(GRID['DIR']) diff --git a/rorapi/management/commands/legacyindexgrid.py b/rorapi/management/commands/legacyindexgrid.py deleted file mode 100644 index 0bb49306..00000000 --- a/rorapi/management/commands/legacyindexgrid.py +++ /dev/null @@ -1,87 +0,0 @@ -import json -import re -import zipfile -from rorapi.settings import ES, ES_VARS, LEGACY_ROR_DUMP - -from django.core.management.base import BaseCommand -from elasticsearch import TransportError - - -def get_nested_names(org): - yield org['name'] - for label in org['labels']: - yield label['label'] - for alias in org['aliases']: - yield alias - for acronym in org['acronyms']: - yield acronym - - -def get_nested_ids(org): - yield org['id'] - yield re.sub('https://', '', org['id']) - yield re.sub('https://ror.org/', '', org['id']) - for ext_name, ext_id in org['external_ids'].items(): - if ext_name == 'GRID': - yield ext_id['all'] - else: - for eid in ext_id['all']: - yield eid - - -class Command(BaseCommand): - help = 'Indexes ROR dataset' - - def handle(self, *args, **options): - with zipfile.ZipFile(LEGACY_ROR_DUMP['ROR_ZIP_PATH'], 'r') as zip_ref: - zip_ref.extractall(LEGACY_ROR_DUMP['DIR']) - - with open(LEGACY_ROR_DUMP['ROR_JSON_PATH'], 'r') as it: - dataset = json.load(it) - - self.stdout.write('Indexing ROR dataset') - - index = ES_VARS['INDEX'] - backup_index = '{}-tmp'.format(index) - ES.reindex(body={ - 'source': { - 'index': index - }, - 'dest': { - 'index': backup_index - } - }) - - try: - for i in range(0, len(dataset), ES_VARS['BULK_SIZE']): - body = [] - for org in dataset[i:i + ES_VARS['BULK_SIZE']]: - body.append({ - 'index': { - '_index': index, - '_type': 'org', - '_id': org['id'] - } - }) - org['names_ids'] = [{ - 'name': n - } for n in get_nested_names(org)] - org['names_ids'] += [{ - 'id': n - } for n in get_nested_ids(org)] - body.append(org) - ES.bulk(body) - except TransportError as e: - self.stdout.write(str(e)) - ES.reindex(body={ - 'source': { - 'index': backup_index - }, - 'dest': { - 'index': index - } - }) - - if ES.indices.exists(backup_index): - ES.indices.delete(backup_index) - self.stdout.write('ROR dataset ' + LEGACY_ROR_DUMP['VERSION'] + ' indexed') diff --git a/rorapi/management/commands/legacyseeschema.py b/rorapi/management/commands/legacyseeschema.py deleted file mode 100644 index 0e016a3a..00000000 --- a/rorapi/management/commands/legacyseeschema.py +++ /dev/null @@ -1,20 +0,0 @@ -import json -from rorapi.settings import ES, ES_VARS - -from django.core.management.base import BaseCommand - - -class Command(BaseCommand): - help = 'Create ROR API index' - - def handle(self, *args, **options): - index = ES_VARS['INDEX'] - if ES.indices.exists(index): - raw_data = ES.indices.get_mapping( index ) - schema = raw_data[ index ]["mappings"]["org"] - print (json.dumps(schema, indent=4)) - else: - with open(ES_VARS['INDEX_TEMPLATE'], 'r') as it: - template = json.load(it) - ES.indices.create(index=index, body=template) - self.stdout.write('Created index {}'.format(index)) diff --git a/rorapi/management/commands/legacyupgrade.py b/rorapi/management/commands/legacyupgrade.py deleted file mode 100644 index 3ea802bb..00000000 --- a/rorapi/management/commands/legacyupgrade.py +++ /dev/null @@ -1,18 +0,0 @@ -from django.core.management.base import BaseCommand -from .downloadgrid import Command as DownloadGridCommand -from .convertgrid import Command as ConvertGridCommand - -# Previously used to generate ROR dataset -# based on the latest GRID dataset configured in settings.py -# As of Mar 2022 ROR is no longer based on GRID -# New records are now created in https://github.com/ror-community/ror-records and pushed to S3 -# Individual record files in S3 are indexed with indexror.py -# Entire dataset zip files in https://github.com/ror-community/ror-data -# can be indexed with setup.py, which uses indexrordump.py - -class Command(BaseCommand): - help = 'Generate up-to-date ror.zip from GRID data' - - def handle(self, *args, **options): - DownloadGridCommand().handle(args, options) - ConvertGridCommand().handle(args, options) diff --git a/rorapi/settings.py b/rorapi/settings.py index dd558a4e..dea38ed7 100644 --- a/rorapi/settings.py +++ b/rorapi/settings.py @@ -55,11 +55,7 @@ # Application definition INSTALLED_APPS = [ - 'django.contrib.admin', - 'django.contrib.auth', 'django.contrib.contenttypes', - 'django.contrib.sessions', - 'django.contrib.messages', 'django.contrib.staticfiles', 'rest_framework', 'django_prometheus', @@ -72,12 +68,7 @@ 'corsheaders.middleware.CorsMiddleware', 'rorapi.middleware.cors.AlwaysAllowOriginMiddleware', 'django.middleware.security.SecurityMiddleware', - 'django.contrib.sessions.middleware.SessionMiddleware', 'django.middleware.common.CommonMiddleware', - 'django.middleware.csrf.CsrfViewMiddleware', - 'django.contrib.auth.middleware.AuthenticationMiddleware', - 'django.contrib.messages.middleware.MessageMiddleware', - 'django.middleware.clickjacking.XFrameOptionsMiddleware', 'django_prometheus.middleware.PrometheusAfterMiddleware' ] @@ -105,6 +96,9 @@ REST_FRAMEWORK = { 'DEFAULT_RENDERER_CLASSES': ('rest_framework.renderers.JSONRenderer', ), + 'DEFAULT_AUTHENTICATION_CLASSES': [], + 'UNAUTHENTICATED_USER': None, + 'UNAUTHENTICATED_TOKEN': None, 'DEFAULT_VERSIONING_CLASS': 'rest_framework.versioning.URLPathVersioning', 'DEFAULT_VERSION': 'v2', 'ALLOWED_VERSIONS': ['v2'], @@ -185,93 +179,11 @@ timeout=240, connection_class=RequestsHttpConnection) -# ROR DUMP grid-2018-11-14 -# GRID = { -# 'VERSION': '2018-11-14', -# 'URL': 'https://ndownloader.figshare.com/files/13575374' -# } - -# ROR DUMP grid-2019-02-17 -# GRID = { -# 'VERSION': '2019-02-17', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/14399291' -# } - -# ROR DUMP ror-2019-09-19 -# GRID = { -# 'VERSION': '2019-05-06', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/15167609' -# } - -# ROR DUMP ror-2019-11-07 -# GRID = { -# 'VERSION': '2019-10-06', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/17948195' -# } - -# ROR DUMP ror-2019-12-18 -# GRID = { -# 'VERSION': '2019-12-10', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/20151785' -# } - -# ROR DUMP ror-2020-03-15 -# GRID = { -# 'VERSION': '2020-03-15', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/22091379' -# } - -# ROR DUMP ror-2020-07-06 -# GRID = { -# 'VERSION': '2020-06-29', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/23552738' -# } - -# ROR DUMP ror-2020-10-19 -#GRID = { -# 'VERSION': '2020-10-06', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/25039403' -# } - -# ROR DUMP 2020-12-21 and 2021-03-17 -#GRID = { -# 'VERSION': '2020-12-09', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/25791104' - -# ROR DUMP 2021-04-06 -#GRID = { -# 'VERSION': '2021-03-25', -# 'URL': 'https://digitalscience.figshare.com/ndownloader/files/27251693' -#} - -# ROR DUMP 2021-09-23 -GRID = { - 'VERSION': '2021-09-16', - 'URL': 'no url' -} -# The latest GRID update, 2021-09-16, was shared via google drive. - -# GRID and LEGACY_ROR_DUMP vars were previously used to -# generate ROR dataset based on the latest GRID dataset -# Directories and files that these vars point to have been moved -# to https://github.com/ror-community/ror-data -# Scripts preprended with 'legacy' no longer work -# As of Mar 2022 ROR is no longer based on GRID -# New records are now created in https://github.com/ror-community/ror-records and pushed to S3 -# Individual record files in S3 are indexed with indexror.py +# As of Mar 2022 ROR is no longer based on GRID. +# New records are created in https://github.com/ror-community/ror-records and pushed to S3. +# Individual record files in S3 are indexed with indexror.py. # Entire dataset zip files in https://github.com/ror-community/ror-data -# can be indexed with setup.py, which uses indexrordump.py - -GRID['DIR'] = os.path.join(BASE_DIR, 'rorapi', 'data', - 'grid-{}'.format(GRID['VERSION'])) -GRID['GRID_ZIP_PATH'] = os.path.join(GRID['DIR'], 'grid.zip') -GRID['GRID_JSON_PATH'] = os.path.join(GRID['DIR'], 'grid.json') - -LEGACY_ROR_DUMP = {'VERSION': '2021-09-23'} -LEGACY_ROR_DUMP['DIR'] = os.path.join(BASE_DIR, 'rorapi', 'data', - 'ror-{}'.format(LEGACY_ROR_DUMP['VERSION'])) -LEGACY_ROR_DUMP['ROR_ZIP_PATH'] = os.path.join(LEGACY_ROR_DUMP['DIR'], 'ror.zip') -LEGACY_ROR_DUMP['ROR_JSON_PATH'] = os.path.join(LEGACY_ROR_DUMP['DIR'], 'ror.json') +# can be indexed with setup.py, which uses indexrordump.py. ROR_DUMP = {} ROR_DUMP['PROD_REPO_URL'] = 'https://api.github.com/repos/ror-community/ror-data' @@ -296,10 +208,6 @@ DATA['DIR'] = os.path.join(BASE_DIR, 'rorapi', 'data') ROR_API = {'PAGE_SIZE': 20, 'ID_PREFIX': 'https://ror.org/'} -GRID_REMOVED_IDS = [] - -LAUNCH_DARKLY_KEY = os.environ.get('LAUNCH_DARKLY_KEY') - # Toggle for behavior-based rate limiting ENABLE_BEHAVIORAL_LIMITING = os.getenv("ENABLE_BEHAVIORAL_LIMITING", "False") == "True" diff --git a/rorapi/tests/tests_integration/tests_search_v2.py b/rorapi/tests/tests_integration/tests_search_v2.py index 462841be..c8279de2 100644 --- a/rorapi/tests/tests_integration/tests_search_v2.py +++ b/rorapi/tests/tests_integration/tests_search_v2.py @@ -133,3 +133,81 @@ def test_extra_word(self): }).json() self.assertTrue(items['number_of_results'] > 0) self.assertEqual(items['items'][0]['id'], 'https://ror.org/00fbnyb24') + + +class CaseAndAccentInsensitiveTestCase(SimpleTestCase): + """Case- and diacritic-insensitive matching (ror-roadmap#175, #398). + + Requires an index created from the current index_template_es7.json. + """ + + def search(self, **params): + return requests.get(BASE_URL, params).json() + + def assert_same_results(self, variants, param='query.advanced'): + results = [self.search(**{param: v}) for v in variants] + baseline = results[0] + self.assertTrue(baseline['number_of_results'] > 0, variants[0]) + baseline_ids = {i['id'] for i in baseline['items']} + for variant, result in zip(variants[1:], results[1:]): + self.assertEqual(result['number_of_results'], + baseline['number_of_results'], variant) + self.assertEqual({i['id'] for i in result['items']}, + baseline_ids, variant) + + def test_keyword_field_case(self): + self.assert_same_results([ + 'locations.geonames_details.name:Denver', + 'locations.geonames_details.name:denver', + 'locations.geonames_details.name:DENVER', + ]) + + def test_relationship_type_case(self): + self.assert_same_results([ + 'relationships.type:child', + 'relationships.type:Child', + ]) + + def test_bare_term_case(self): + self.assert_same_results(['Stellenbosch', 'stellenbosch']) + + def test_keyword_field_diacritics(self): + self.assert_same_results([ + 'locations.geonames_details.name:Huế', + 'locations.geonames_details.name:Hue', + 'locations.geonames_details.name:hue', + ]) + self.assert_same_results([ + 'locations.geonames_details.name:Montréal', + 'locations.geonames_details.name:Montreal', + ]) + + def test_nfd_and_nfc_input(self): + nfc = 'locations.geonames_details.name:Hu\u1ebf' + nfd = 'locations.geonames_details.name:Hu\u0065\u0302\u0301' + self.assertNotEqual(nfc, nfd) + self.assert_same_results([nfc, nfd]) + + def test_text_field_diacritics(self): + self.assert_same_results([ + 'names.value:"Université de Montréal"', + 'names.value:"Universite de Montreal"', + 'names.value:"universite de montreal"', + ]) + + def test_filter_case(self): + self.assert_same_results(['Denver'], param='query') + upper = requests.get(BASE_URL, {'filter': 'types:Education'}).json() + lower = requests.get(BASE_URL, {'filter': 'types:education'}).json() + self.assertTrue(upper['number_of_results'] > 0) + self.assertEqual(upper['number_of_results'], + lower['number_of_results']) + + def test_aggregation_keys_keep_original_casing(self): + # Bucket titles are derived from the raw (un-normalized) keys, e.g. the + # country name is looked up from the upper-case ISO code. + meta = self.search(**{'query.advanced': 'locations.geonames_details.name:denver'})['meta'] + countries = {b['id']: b['title'] for b in meta['countries']} + self.assertEqual(countries.get('us'), 'United States') + continents = {b['id']: b['title'] for b in meta['continents']} + self.assertEqual(continents.get('na'), 'North America') diff --git a/rorapi/tests/tests_unit/tests_queries_v2.py b/rorapi/tests/tests_unit/tests_queries_v2.py index 130ff931..251b5424 100644 --- a/rorapi/tests/tests_unit/tests_queries_v2.py +++ b/rorapi/tests/tests_unit/tests_queries_v2.py @@ -168,10 +168,10 @@ class BuildSearchQueryTestCase(SimpleTestCase): def setUp(self): self.default_query = \ - {'aggs': {'types': {'terms': {'field': 'types', 'size': 10, 'min_doc_count': 1}}, - 'countries': {'terms': {'field': 'locations.geonames_details.country_code', 'size': 10, 'min_doc_count': 1}}, - 'continents': {'terms': {'field': 'locations.geonames_details.continent_code', 'size': 10, 'min_doc_count': 1}}, - 'statuses': {'terms': {'field': 'status', 'size': 10, 'min_doc_count': 1}}}, + {'aggs': {'types': {'terms': {'field': 'types.raw', 'size': 10, 'min_doc_count': 1}}, + 'countries': {'terms': {'field': 'locations.geonames_details.country_code.raw', 'size': 10, 'min_doc_count': 1}}, + 'continents': {'terms': {'field': 'locations.geonames_details.continent_code.raw', 'size': 10, 'min_doc_count': 1}}, + 'statuses': {'terms': {'field': 'status.raw', 'size': 10, 'min_doc_count': 1}}}, 'track_total_hits': True, 'from': 0, 'size': 20} def test_empty_query_default(self): @@ -284,6 +284,38 @@ def test_query_advanced(self): query = build_search_query({'query.advanced': 'query terms'}) self.assertEqual(query.to_dict(), expected) + def test_query_advanced_nfc_normalization(self): + nfd = 'locations.geonames_details.name:Hu\u0065\u0302\u0301' + nfc_form = 'locations.geonames_details.name:Hu\u1ebf' + self.assertNotEqual(nfd, nfc_form) + query = build_search_query({'query.advanced': nfd}).to_dict() + self.assertEqual( + query['query']['bool']['must'][0]['query_string']['query'], nfc_form) + + def test_query_nfc_normalization(self): + query = build_search_query({'query': 'Montre\u0301al'}).to_dict() + self.assertEqual( + query['query']['bool']['must'][0]['nested']['query']['query_string']['query'], + 'Montr\u00e9al') + + def test_filter_nfc_normalization(self): + query = build_search_query({ + 'filter': 'locations.geonames_details.country_name:Cura\u0063\u0327ao', + 'all_status': '' + }).to_dict() + self.assertIn( + {'terms': {'locations.geonames_details.country_name': ('Cura\u00e7ao',)}}, + query['query']['bool']['filter']) + + def test_aggregations_use_raw_subfields(self): + aggs = build_search_query({}).to_dict()['aggs'] + self.assertEqual(aggs['types']['terms']['field'], 'types.raw') + self.assertEqual(aggs['countries']['terms']['field'], + 'locations.geonames_details.country_code.raw') + self.assertEqual(aggs['continents']['terms']['field'], + 'locations.geonames_details.continent_code.raw') + self.assertEqual(aggs['statuses']['terms']['field'], 'status.raw') + def test_query_advanced_all_status(self): expected = {'query': { 'bool': { diff --git a/rorapi/tests/tests_unit/tests_register_throttle.py b/rorapi/tests/tests_unit/tests_register_throttle.py new file mode 100644 index 00000000..ca3d5c4f --- /dev/null +++ b/rorapi/tests/tests_unit/tests_register_throttle.py @@ -0,0 +1,29 @@ +from django.core.cache import cache +from django.test import SimpleTestCase +from rest_framework.exceptions import Throttled +from rest_framework.test import APIRequestFactory + +from rorapi.common.views import ClientRegistrationThrottle, ClientRegistrationView + +factory = APIRequestFactory() + + +class ClientRegistrationThrottleTests(SimpleTestCase): + def setUp(self): + # DRF SimpleRateThrottle binds django.core.cache.cache at import time. + ClientRegistrationThrottle.cache = cache + cache.clear() + self.view = ClientRegistrationView() + + def test_view_uses_class_rate_throttle(self): + self.assertEqual(ClientRegistrationThrottle.rate, "5/hour") + self.assertEqual( + ClientRegistrationView.throttle_classes, [ClientRegistrationThrottle] + ) + + def test_allows_five_anonymous_requests_then_throttles(self): + request = self.view.initialize_request(factory.post("/v2/register")) + for _ in range(5): + self.view.check_throttles(request) + with self.assertRaises(Throttled): + self.view.check_throttles(request) diff --git a/rorapi/v2/index_template_es7.json b/rorapi/v2/index_template_es7.json index a16c10db..152ab18d 100644 --- a/rorapi/v2/index_template_es7.json +++ b/rorapi/v2/index_template_es7.json @@ -19,6 +19,15 @@ "type": "asciifolding", "preserve_original": true } + }, + "normalizer": { + "folding_normalizer": { + "type": "custom", + "filter": [ + "lowercase", + "asciifolding" + ] + } } } }, @@ -61,7 +70,8 @@ "type": "keyword" }, "type": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "preferred": { "type": "keyword" @@ -78,7 +88,8 @@ "analyzer": "simple" }, "type": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" } } }, @@ -90,22 +101,38 @@ "geonames_details": { "properties": { "continent_code": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer", + "fields": { + "raw": { + "type": "keyword" + } + } }, "continent_name": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "country_code": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer", + "fields": { + "raw": { + "type": "keyword" + } + } }, "country_name": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "country_subdivision_code": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "country_subdivision_name": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "lat": { "type": "float" @@ -114,7 +141,8 @@ "type": "float" }, "name": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" } } } @@ -133,23 +161,33 @@ "analyzer": "string_lowercase", "fielddata": true } - } + }, + "analyzer": "string_lowercase" }, "lang": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "types": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" } } }, "types": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer", + "fields": { + "raw": { + "type": "keyword" + } + } }, "relationships": { "properties": { "type": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer" }, "label": { "type": "text", @@ -162,7 +200,8 @@ "analyzer": "string_lowercase", "fielddata": true } - } + }, + "analyzer": "string_lowercase" }, "id": { "type": "keyword" @@ -170,7 +209,13 @@ } }, "status": { - "type": "keyword" + "type": "keyword", + "normalizer": "folding_normalizer", + "fields": { + "raw": { + "type": "keyword" + } + } }, "names_ids": { "type": "nested", @@ -223,4 +268,4 @@ } } } -} \ No newline at end of file +} diff --git a/rorapi/v2/models.py b/rorapi/v2/models.py index b937d7dd..9cfdb136 100644 --- a/rorapi/v2/models.py +++ b/rorapi/v2/models.py @@ -1,4 +1,3 @@ -from geonamescache.mappers import country import random import string from django.db import models @@ -13,19 +12,6 @@ def __init__(self, data): self.title = continent_code_to_name(data.key) self.count = data.doc_count -class CountryBucket: - """A model class for country aggregation bucket""" - - def __init__(self, data): - self.id = data.key.lower() - mapper = country(from_key="iso", to_key="name") - try: - self.title = mapper(data.key) - except AttributeError: - # if we have a country code with no name mapping, skip it to prevent 500 - pass - self.count = data.doc_count - class Aggregations: """Aggregations model class""" diff --git a/rorapi/v2/tests.py b/rorapi/v2/tests.py deleted file mode 100644 index 02b18058..00000000 --- a/rorapi/v2/tests.py +++ /dev/null @@ -1,12 +0,0 @@ -from django.test import TestCase -from rorapi.v2.models import Client - -class ClientTests(TestCase): - def test_client_registration(self): - client = Client.objects.create(email='test@example.com') - self.assertIsNotNone(client.client_id) - - def test_validate_client_id(self): - response = self.client.get('/validate-client-id/INVALID_ID/') - self.assertEqual(response.status_code, 200) - self.assertFalse(response.json()['valid'])