From 73a96e11f8b13b6ed9c350161a74faabeafec011 Mon Sep 17 00:00:00 2001 From: Joseph Rhoads Date: Fri, 2 Oct 2026 09:14:40 +0200 Subject: [PATCH 1/2] Run unit tests without the Elasticsearch data load (#599) Co-authored-by: Cursor Agent --- .github/workflows/run_tests.yml | 15 +++++++++++---- rorapi/tests/tests_integration/tests.py | 6 ++++++ rorapi/tests/tests_integration/tests_heartbeat.py | 10 ++++++++++ rorapi/tests/tests_unit/tests_views_v2.py | 14 -------------- 4 files changed, 27 insertions(+), 18 deletions(-) create mode 100644 rorapi/tests/tests_integration/tests_heartbeat.py diff --git a/.github/workflows/run_tests.yml b/.github/workflows/run_tests.yml index 392cabd..9788738 100644 --- a/.github/workflows/run_tests.yml +++ b/.github/workflows/run_tests.yml @@ -60,14 +60,21 @@ jobs: run: | python -Wd manage.py check python manage.py makemigrations --check --dry-run + - name: Run unit tests + working-directory: ./ror-api + run: | + python -Wa manage.py test rorapi.tests.tests_unit - name: Load test data working-directory: ./ror-api run: | python manage.py setup ${{ env.TEST_DATA_DUMP_FILE }} -t - - name: Run Tests + - name: Run integration tests working-directory: ./ror-api run: | - python -Wa manage.py test rorapi.tests.tests_unit - # TODO fix these tests running in GitHub Action - # python manage.py test rorapi.tests.tests_integration + python manage.py test rorapi.tests.tests_integration.tests_heartbeat + # HTTP suites assume a full public corpus (specific ROR IDs, >50k + # records). The CI test dump does not satisfy that, so they stay off. + # python manage.py test rorapi.tests.tests_integration.tests_v2 + # python manage.py test rorapi.tests.tests_integration.tests_search_v2 + # python manage.py test rorapi.tests.tests_integration.tests_matching_v2 # python manage.py test rorapi.tests.tests_functional diff --git a/rorapi/tests/tests_integration/tests.py b/rorapi/tests/tests_integration/tests.py index 7336414..2c8ec62 100644 --- a/rorapi/tests/tests_integration/tests.py +++ b/rorapi/tests/tests_integration/tests.py @@ -2,6 +2,8 @@ import json import os import re +import unittest + import requests from django.test import SimpleTestCase @@ -11,6 +13,10 @@ os.environ.get('ROR_BASE_URL', 'http://localhost')) +@unittest.skip( + 'Expects the removed v1 response shape (name, country.country_code). ' + 'v2 coverage is tests_v2.py.' +) class APITestCase(SimpleTestCase): def get_total(self, output): return output['number_of_results'] diff --git a/rorapi/tests/tests_integration/tests_heartbeat.py b/rorapi/tests/tests_integration/tests_heartbeat.py new file mode 100644 index 0000000..29869c5 --- /dev/null +++ b/rorapi/tests/tests_integration/tests_heartbeat.py @@ -0,0 +1,10 @@ +from django.test import SimpleTestCase + + +class HeartbeatViewTestCase(SimpleTestCase): + """Needs a live organizations-v2 index, so it runs with the integration suite.""" + + def test_heartbeat_success(self): + response = self.client.get('/heartbeat') + self.assertEqual(response.status_code, 200) + self.assertEqual(response.content, b'OK') diff --git a/rorapi/tests/tests_unit/tests_views_v2.py b/rorapi/tests/tests_unit/tests_views_v2.py index 5816532..e046f44 100644 --- a/rorapi/tests/tests_unit/tests_views_v2.py +++ b/rorapi/tests/tests_unit/tests_views_v2.py @@ -389,20 +389,6 @@ def test_index_ror_success_with_auth_headers(self, index_mock): ) self.assertEqual(response.status_code, 200) -class HeartbeatViewTestCase(SimpleTestCase): - def setUp(self): - with open( - os.path.join(os.path.dirname(__file__), - 'data/test_data_search_es7_v2.json'), 'r') as f: - self.test_data = json.load(f) - - @mock.patch('elasticsearch_dsl.Search.execute') - def test_heartbeat_success(self, search_mock): - search_mock.return_value = \ - IterableAttrDict(self.test_data, self.test_data['hits']['hits']) - response = self.client.get('/v2/heartbeat') - self.assertEqual(response.status_code, 200) - class BulkUpdateViewTestCase(SimpleTestCase): def setUp(self): self.csv_errors_empty = [] From 97a20efec7834f195fe220ee6d5b1b95051c7c15 Mon Sep 17 00:00:00 2001 From: Joseph Rhoads Date: Fri, 2 Oct 2026 09:49:03 +0200 Subject: [PATCH 2/2] Exclude acronyms from multisearch candidate retrieval (#600) Co-authored-by: Cursor Agent --- rorapi/common/es_utils.py | 41 ++++++++++++ rorapi/common/index_helpers.py | 3 + rorapi/common/matching.py | 12 ++-- .../tests_integration/tests_matching_v2.py | 18 +++++ rorapi/tests/tests_unit/tests_es_utils_v2.py | 66 +++++++++++++++++++ .../tests/tests_unit/tests_index_helpers.py | 1 + rorapi/v2/index_template_es7.json | 4 ++ 7 files changed, 139 insertions(+), 6 deletions(-) diff --git a/rorapi/common/es_utils.py b/rorapi/common/es_utils.py index e1ff9cd..449e0fd 100644 --- a/rorapi/common/es_utils.py +++ b/rorapi/common/es_utils.py @@ -70,6 +70,47 @@ def add_common_query(self, fields, terms): ], ) + def add_phrase_query_affiliation_names(self, terms): + """Phrase match on acronym-free names in ``affiliation_match.names``.""" + self.search.query = Q( + "nested", + path="affiliation_match.names", + score_mode="max", + query=Q("match_phrase", **{"affiliation_match.names.name": terms}), + ) + + def add_common_query_affiliation_names(self, terms): + self.search.query = Q( + "nested", + path="affiliation_match.names", + score_mode="max", + query=Q( + "common", + **{ + "affiliation_match.names.name": { + "query": terms, + "cutoff_frequency": 0.001, + } + }, + ), + ) + + def add_fuzzy_query_affiliation_names(self, terms): + self.search.query = Q( + "nested", + path="affiliation_match.names", + score_mode="max", + query=Q( + "match", + **{ + "affiliation_match.names.name": { + "query": terms, + "fuzziness": "AUTO", + } + }, + ), + ) + def add_match_query(self, terms): self.search = self.search.query("match", acronyms=terms) diff --git a/rorapi/common/index_helpers.py b/rorapi/common/index_helpers.py index 2ee8c65..239ace9 100644 --- a/rorapi/common/index_helpers.py +++ b/rorapi/common/index_helpers.py @@ -53,6 +53,9 @@ def enrich_org_for_index(org): 'id': n } for n in get_nested_ids_v2(org)] org['affiliation_match'] = get_affiliation_match_doc(org) + org['acronyms'] = [ + n["value"] for n in org["names"] if "acronym" in n["types"] + ] return org diff --git a/rorapi/common/matching.py b/rorapi/common/matching.py index 2ef74c9..228fe4a 100644 --- a/rorapi/common/matching.py +++ b/rorapi/common/matching.py @@ -179,7 +179,6 @@ def match_by_query(text, matching_type, query, countries): def match_by_type(text, matching_type, countries): """Match affiliation text using specific matching mode/type.""" - fields = ["names.value.norm"] substrings = [] if matching_type == MATCHING_TYPE_HEURISTICS: h1 = re.search(r"University of ([^\s]+)", text) @@ -204,17 +203,18 @@ def match_by_type(text, matching_type, countries): queries = [ESQueryBuilder() for _ in substrings] + normalized = normalize(text) for s, q in zip(substrings, queries): if matching_type == MATCHING_TYPE_PHRASE: - q.add_phrase_query(fields, normalize(text)) + q.add_phrase_query_affiliation_names(normalized) elif matching_type == MATCHING_TYPE_COMMON: - q.add_common_query(fields, normalize(text)) + q.add_common_query_affiliation_names(normalized) elif matching_type == MATCHING_TYPE_FUZZY: - q.add_fuzzy_query(fields, normalize(text)) + q.add_fuzzy_query_affiliation_names(normalized) elif matching_type == MATCHING_TYPE_ACRONYM: - q.add_match_query(normalize(text)) + q.add_match_query(normalized) elif matching_type == MATCHING_TYPE_HEURISTICS: - q.add_common_query(fields, normalize(text)) + q.add_common_query_affiliation_names(normalized) queries = [q.get_query() for q in queries] matched = [ match_by_query(t, matching_type, q, countries) diff --git a/rorapi/tests/tests_integration/tests_matching_v2.py b/rorapi/tests/tests_integration/tests_matching_v2.py index 7f64cd7..1cc2d23 100644 --- a/rorapi/tests/tests_integration/tests_matching_v2.py +++ b/rorapi/tests/tests_integration/tests_matching_v2.py @@ -41,3 +41,21 @@ def test_query_organizations(self): self.assertTrue( i.get('matching_type') in ['PHRASE', 'ACRONYM', 'FUZZY', 'HEURISTICS', 'COMMON TERMS', 'EXACT']) + + +class MultisearchAcronymExclusionTestCase(SimpleTestCase): + """ror-roadmap#345: multisearch must not choose orgs via acronym-only names.""" + + def test_ucla_does_not_choose_wrong_org(self): + output = requests.get(BASE_URL, { + 'affiliation': 'UCLA', + 'multisearch': '', + }).json() + chosen = [i for i in output.get('items', []) if i.get('chosen')] + for item in chosen: + self.assertNotEqual( + item['organization']['id'], + 'https://ror.org/03qgg3111', + 'Universidad Centroccidental Lisandro Alvarado must not be chosen ' + 'for bare UCLA via PHRASE on acronym', + ) diff --git a/rorapi/tests/tests_unit/tests_es_utils_v2.py b/rorapi/tests/tests_unit/tests_es_utils_v2.py index ec00379..6828abf 100644 --- a/rorapi/tests/tests_unit/tests_es_utils_v2.py +++ b/rorapi/tests/tests_unit/tests_es_utils_v2.py @@ -161,6 +161,72 @@ def test_fuzzy_query(self): 'track_total_hits': True }) + def test_phrase_query_affiliation_names(self): + qb = ESQueryBuilder() + qb.add_phrase_query_affiliation_names('query terms') + + self.assertEqual( + qb.get_query().to_dict(), { + 'query': { + 'nested': { + 'path': 'affiliation_match.names', + 'score_mode': 'max', + 'query': { + 'match_phrase': { + 'affiliation_match.names.name': 'query terms' + } + } + } + }, + 'track_total_hits': True + }) + + def test_common_query_affiliation_names(self): + qb = ESQueryBuilder() + qb.add_common_query_affiliation_names('query terms') + + self.assertEqual( + qb.get_query().to_dict(), { + 'query': { + 'nested': { + 'path': 'affiliation_match.names', + 'score_mode': 'max', + 'query': { + 'common': { + 'affiliation_match.names.name': { + 'query': 'query terms', + 'cutoff_frequency': 0.001 + } + } + } + } + }, + 'track_total_hits': True + }) + + def test_fuzzy_query_affiliation_names(self): + qb = ESQueryBuilder() + qb.add_fuzzy_query_affiliation_names('query terms') + + self.assertEqual( + qb.get_query().to_dict(), { + 'query': { + 'nested': { + 'path': 'affiliation_match.names', + 'score_mode': 'max', + 'query': { + 'match': { + 'affiliation_match.names.name': { + 'query': 'query terms', + 'fuzziness': 'AUTO' + } + } + } + } + }, + 'track_total_hits': True + }) + def test_add_filters(self): qb = ESQueryBuilder() qb.add_match_all_query() diff --git a/rorapi/tests/tests_unit/tests_index_helpers.py b/rorapi/tests/tests_unit/tests_index_helpers.py index e0b728d..ce09e65 100644 --- a/rorapi/tests/tests_unit/tests_index_helpers.py +++ b/rorapi/tests/tests_unit/tests_index_helpers.py @@ -73,6 +73,7 @@ def test_enrich_org_for_index(self): self.assertIn({'name': 'University of Example'}, org['names_ids']) self.assertIn({'id': '01an7q238'}, org['names_ids']) self.assertEqual(org['affiliation_match']['primary'], 'University of Example') + self.assertEqual(org['acronyms'], ['UoE']) class BulkIndexWithBackupTestCase(SimpleTestCase): diff --git a/rorapi/v2/index_template_es7.json b/rorapi/v2/index_template_es7.json index 152ab18..bb5a450 100644 --- a/rorapi/v2/index_template_es7.json +++ b/rorapi/v2/index_template_es7.json @@ -61,6 +61,10 @@ "type": "text", "analyzer": "simple" }, + "acronyms": { + "type": "keyword", + "normalizer": "folding_normalizer" + }, "established": { "type": "date" },