From f54c47844e55be774acb7c74cb7083abf568867e Mon Sep 17 00:00:00 2001 From: Saksham Date: Fri, 26 Jun 2026 12:16:33 +0200 Subject: [PATCH 1/4] feat(rdm.migration): Configurable auto-adding secondary community --- cds_migrator_kit/errors.py | 6 + .../rdm/records/transform/config.py | 16 ++ .../rdm/records/transform/entities/parent.py | 25 +++ .../rdm/records/transform/transform.py | 2 + tests/cds-rdm/conftest.py | 14 +- tests/cds-rdm/test_bulletin_issue.py | 5 - tests/cds-rdm/test_ep_approval_entry.py | 7 +- tests/cds-rdm/test_full_migration.py | 9 ++ tests/cds-rdm/test_scientific_community.py | 147 ++++++++++++++++++ 9 files changed, 222 insertions(+), 9 deletions(-) create mode 100644 tests/cds-rdm/test_scientific_community.py diff --git a/cds_migrator_kit/errors.py b/cds_migrator_kit/errors.py index 5cbe912d..a998ac4a 100644 --- a/cds_migrator_kit/errors.py +++ b/cds_migrator_kit/errors.py @@ -85,3 +85,9 @@ class RecordFlaggedCuration(CDSMigrationException): """Record statistics error.""" description = "[Record needs to be curated]" + + +class MissingConfiguration(CDSMigrationException): + """Missing configuration exception.""" + + description = "[Missing configuration]" diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 2bca4a47..1936f099 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -163,3 +163,19 @@ EXPERIMENT_ALIASES = { "t2k": "re13", } + +# Public research publication resource types that are auto-included in the CERN Research community. +CDS_CERN_SCIENTIFIC_RESOURCE_TYPES = { + "publication-dissertation", # Already included by the migrator for thesis records. + "publication-book", + "publication-section", + "publication-conferencepaper", + "publication-conferenceproceeding", + "publication-journal", + "publication-article", + "publication-preprint", + "publication-report", + "publication-technicalnote", + "publication-note", + "publication", +} diff --git a/cds_migrator_kit/rdm/records/transform/entities/parent.py b/cds_migrator_kit/rdm/records/transform/entities/parent.py index adfd130a..5dfcc684 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/parent.py +++ b/cds_migrator_kit/rdm/records/transform/entities/parent.py @@ -16,6 +16,8 @@ from cds_migrator_kit.errors import ManualImportRequired, UnexpectedValue from cds_migrator_kit.rdm.records.transform.mappers.base import RecordTransformContext from cds_migrator_kit.rdm.records.transform.mappers.record import AccessGrantsMapper +from cds_migrator_kit.rdm.records.transform.config import CDS_CERN_SCIENTIFIC_RESOURCE_TYPES +from cds_migrator_kit.errors import MissingConfiguration EMAIL_PATTERN = re.compile(r"[^@]+@[^@]+\.[^@]+") @@ -98,10 +100,33 @@ def _build_access(self): ) return {"owned_by": {"user": owner}} + def _should_add_scientific_community(self): + if self.restricted or self.record.access_status != "public": + return False + resource_type_id = ( + self.record.body + .get("metadata", {}) + .get("resource_type", {}) + .get("id") + ) + return resource_type_id in CDS_CERN_SCIENTIFIC_RESOURCE_TYPES + def _build_communities(self): """Combine the configured target communities with the record's own.""" communities = self.dojson_entry.pop("communities", []) communities = self.communities_ids + [slug for slug in communities] + + scientific_community = current_app.config.get( + "CDS_CERN_SCIENTIFIC_COMMUNITY_ID" + ) + if not scientific_community: + raise MissingConfiguration( + "CDS_CERN_SCIENTIFIC_COMMUNITY_ID is not configured" + ) + if self._should_add_scientific_community(): + if scientific_community not in communities: + communities.append(scientific_community) + if communities: return {"ids": communities, "default": self.communities_ids[0]} return {} diff --git a/cds_migrator_kit/rdm/records/transform/transform.py b/cds_migrator_kit/rdm/records/transform/transform.py index bcb60b1b..8c362175 100644 --- a/cds_migrator_kit/rdm/records/transform/transform.py +++ b/cds_migrator_kit/rdm/records/transform/transform.py @@ -27,6 +27,7 @@ from cds_migrator_kit.errors import ( ManualImportRequired, + MissingConfiguration, MissingRequiredField, MultipleModelsMatched, RestrictedFileDetected, @@ -208,6 +209,7 @@ def _transform(self, raw_dump_entry) -> Optional[MigrationEntry]: ManualImportRequired, MissingRequiredField, MultipleModelsMatched, + MissingConfiguration, ) as e: migration_logger.add_log(e, record=raw_dump_entry) diff --git a/tests/cds-rdm/conftest.py b/tests/cds-rdm/conftest.py index 8cf329e7..41bbead0 100644 --- a/tests/cds-rdm/conftest.py +++ b/tests/cds-rdm/conftest.py @@ -278,7 +278,7 @@ def running_app( @pytest.fixture -def test_app(running_app): +def test_app(running_app, cern_scientific_community): """Get current app.""" running_app.app.config["RDM_PERSISTENT_IDENTIFIERS"]["doi"]["required"] = False running_app.app.config["RDM_PARENT_PERSISTENT_IDENTIFIERS"]["doi"][ @@ -1723,6 +1723,18 @@ def community(running_app, db): return comm +@pytest.fixture() +def cern_scientific_community(running_app, db, app): + """A CERN Research community fixture.""" + comm = Community.create({}) + comm.slug = "cern-research" + comm.metadata = {"title": "CERN Research"} + comm.commit() + db.session.commit() + app.config["CDS_CERN_SCIENTIFIC_COMMUNITY_ID"] = str(comm.id) + return comm + + # @pytest.fixture() # def users(app, db): # """Create example user.""" diff --git a/tests/cds-rdm/test_bulletin_issue.py b/tests/cds-rdm/test_bulletin_issue.py index 88cd0bdf..12c2d0f9 100644 --- a/tests/cds-rdm/test_bulletin_issue.py +++ b/tests/cds-rdm/test_bulletin_issue.py @@ -80,8 +80,3 @@ def test_bulletin_issue( ) if record["legacy_recid"] == "2234683": parent_related_identifier(loaded_rec) - - -# FAILED tests/cds-rdm/test_access_permissions.py::test_access_permissions - assert 2 == 3 -# FAILED tests/cds-rdm/test_bulletin_issue.py::test_bulletin_issue - AssertionError: assert [{'identifier...heme': 'cds'}] == [{'identifier...heme': 'cds'}] -# FAILED tests/cds-rdm/test_full_migration.py::test_full_migration_stream - assert 0 == 2 diff --git a/tests/cds-rdm/test_ep_approval_entry.py b/tests/cds-rdm/test_ep_approval_entry.py index 3cb97593..d028edd9 100644 --- a/tests/cds-rdm/test_ep_approval_entry.py +++ b/tests/cds-rdm/test_ep_approval_entry.py @@ -19,6 +19,7 @@ EPPHAPP_FILE_TYPE, PublicEntry, RestrictedEntry, + _cern_scientific_community_id, ) from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent from cds_migrator_kit.rdm.records.transform.entities.record import RecordEntry @@ -568,20 +569,20 @@ def test_public_adds_cern_scientific_community(self, app): entry, _make_approval_request(), _make_migration_logger() ).build() - assert CDS_CERN_SCIENTIFIC_COMMUNITY_ID in result["parent"].communities["ids"] + assert _cern_scientific_community_id() in result["parent"].communities["ids"] def test_public_does_not_duplicate_community(self, app): entry = _make_entry(_versions_with_epphapp()) entry["parent"].communities["ids"] = [ "example-community", - CDS_CERN_SCIENTIFIC_COMMUNITY_ID, + _cern_scientific_community_id(), ] result = PublicEntry( entry, _make_approval_request(), _make_migration_logger() ).build() community_ids = result["parent"].communities["ids"] - assert community_ids.count(CDS_CERN_SCIENTIFIC_COMMUNITY_ID) == 1 + assert community_ids.count(_cern_scientific_community_id()) == 1 class TestEntryImmutability: diff --git a/tests/cds-rdm/test_full_migration.py b/tests/cds-rdm/test_full_migration.py index ee45fa23..6e0e9cfd 100644 --- a/tests/cds-rdm/test_full_migration.py +++ b/tests/cds-rdm/test_full_migration.py @@ -482,6 +482,7 @@ def test_full_migration_stream( superuser_identity, orcid_name_data, community, + cern_scientific_community, # Fixture for the CERN Scientific community, so that it gets auto included for PUBLIC-ations. mocker, groups, ): @@ -537,6 +538,8 @@ def test_full_migration_stream( if record["legacy_recid"] == "2294138": author_with_inspire(loaded_rec) + # Check if the record is also included in the CERN Scientific community since it is a publication report + scientific_community_inclusion(loaded_rec, cern_scientific_community.id) # Check if remote account has the correct metadata # Check if user profile has the correct metadata user_metadata() @@ -606,3 +609,9 @@ def user_metadata(): "person_id": "11115", "department": "IT", } + + +def scientific_community_inclusion(record, cern_scientific_community_uuid): + """Checks if the record is included in the CERN Scientific community.""" + assert str(cern_scientific_community_uuid) in record._record.parent.communities.ids + assert cern_scientific_community_uuid != record._record.parent.communities.default diff --git a/tests/cds-rdm/test_scientific_community.py b/tests/cds-rdm/test_scientific_community.py new file mode 100644 index 00000000..e7b79ab4 --- /dev/null +++ b/tests/cds-rdm/test_scientific_community.py @@ -0,0 +1,147 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for auto-inclusion in the CERN Research community.""" + +from unittest.mock import MagicMock + +import pytest + +from cds_migrator_kit.errors import MissingConfiguration +from cds_migrator_kit.rdm.records.transform.config import ( + CDS_CERN_SCIENTIFIC_RESOURCE_TYPES, +) +from cds_migrator_kit.rdm.records.transform.transform import CDSToRDMRecordTransform + + +def _test_record( + access="public", + resource_type="publication-preprint", + communities=[], + recid="123456", +): + """Build a minimal CDSToRDMRecordEntry.transform() output for community tests.""" + metadata = { + "title": "Test record", + "publication_date": "2020-01-01", + } + if resource_type is not None: + metadata["resource_type"] = {"id": resource_type} + + return { + "recid": recid, + "access": access, + "communities": communities, + "json": { + "files": {"enabled": False}, + "metadata": metadata, + }, + } + + +@pytest.fixture +def transform(tmp_path, community): + """Transform instance with a collection community configured.""" + return CDSToRDMRecordTransform( + files_dump_dir=tmp_path, + missing_users=tmp_path, + communities_ids=[str(community.id)], + migration_logger=MagicMock(), + ) + + +class TestCommunitiesIds: + """Test CDSToRDMRecordTransform._communities_ids().""" + + def test_adds_scientific_community_for_public_research_test_record( + self, transform, community, cern_scientific_community + ): + """Public research records are included in the CERN Scientific community.""" + result = transform._communities_ids({}, _test_record()) + + assert result == { + "ids": [str(community.id), str(cern_scientific_community.id)], + "default": str(community.id), + } + + def test_keep_collection_community_as_default( + self, transform, community, cern_scientific_community + ): + """Collection community remains the default when CERN Scientific community is added.""" + result = transform._communities_ids( + {}, + _test_record(communities=["test-community"]), + ) + + assert result["default"] == str(community.id) + assert result["ids"] == [ + str(community.id), + "test-community", + str(cern_scientific_community.id), + ] + + @pytest.mark.parametrize("resource_type", CDS_CERN_SCIENTIFIC_RESOURCE_TYPES) + def test_research_resource_types( + self, transform, cern_scientific_community, resource_type + ): + """All configured public research resource types trigger inclusion.""" + result = transform._communities_ids( + {}, _test_record(resource_type=resource_type) + ) + + assert str(cern_scientific_community.id) in result["ids"] + assert cern_scientific_community.id != result["default"] + + def test_skip_restricted_test_record( + self, transform, community, cern_scientific_community + ): + """Restricted records are not included in the CERN Research community.""" + result = transform._communities_ids({}, _test_record(access="restricted")) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_non_research_resource_type( + self, transform, community, cern_scientific_community + ): + """Non-research resource types are not included in the CERN Scientific community.""" + result = transform._communities_ids({}, _test_record(resource_type="other")) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_when_stream_is_restricted( + self, tmp_path, community, cern_scientific_community + ): + """Records on restricted migration streams are not included in the CERN Scientific community.""" + transform = CDSToRDMRecordTransform( + files_dump_dir=tmp_path, + missing_users=tmp_path, + communities_ids=[str(community.id)], + restricted=True, + migration_logger=MagicMock(), + ) + + result = transform._communities_ids({}, _test_record()) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_raise_when_cern_scientific_community_not_configured( + self, test_app, transform, community, monkeypatch + ): + """No CERN Scientific community is added when config is unset.""" + monkeypatch.setitem(test_app.config, "CDS_CERN_SCIENTIFIC_COMMUNITY_ID", None) + + with pytest.raises(MissingConfiguration): + transform._communities_ids({}, _test_record()) From b606043ab5ac73fe10171e8aa9f6160518bd7097 Mon Sep 17 00:00:00 2001 From: Saksham Date: Fri, 31 Jul 2026 17:46:08 +0200 Subject: [PATCH 2/4] rdm.transform: don't auto add community if files are restricted --- .../rdm/records/transform/config.py | 2 +- .../rdm/records/transform/entities/parent.py | 6 ++- tests/cds-rdm/test_full_migration.py | 2 + tests/cds-rdm/test_scientific_community.py | 46 +++++++++++++++---- 4 files changed, 44 insertions(+), 12 deletions(-) diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 1936f099..fdbb52c2 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -165,7 +165,7 @@ } # Public research publication resource types that are auto-included in the CERN Research community. -CDS_CERN_SCIENTIFIC_RESOURCE_TYPES = { +CERN_SCIENTIFIC_RESOURCE_TYPES = { "publication-dissertation", # Already included by the migrator for thesis records. "publication-book", "publication-section", diff --git a/cds_migrator_kit/rdm/records/transform/entities/parent.py b/cds_migrator_kit/rdm/records/transform/entities/parent.py index 5dfcc684..020227c4 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/parent.py +++ b/cds_migrator_kit/rdm/records/transform/entities/parent.py @@ -16,7 +16,7 @@ from cds_migrator_kit.errors import ManualImportRequired, UnexpectedValue from cds_migrator_kit.rdm.records.transform.mappers.base import RecordTransformContext from cds_migrator_kit.rdm.records.transform.mappers.record import AccessGrantsMapper -from cds_migrator_kit.rdm.records.transform.config import CDS_CERN_SCIENTIFIC_RESOURCE_TYPES +from cds_migrator_kit.rdm.records.transform.config import CERN_SCIENTIFIC_RESOURCE_TYPES from cds_migrator_kit.errors import MissingConfiguration EMAIL_PATTERN = re.compile(r"[^@]+@[^@]+\.[^@]+") @@ -103,13 +103,15 @@ def _build_access(self): def _should_add_scientific_community(self): if self.restricted or self.record.access_status != "public": return False + if any(file.get("status") for file in self.dojson_entry.get("files", [])): + return False resource_type_id = ( self.record.body .get("metadata", {}) .get("resource_type", {}) .get("id") ) - return resource_type_id in CDS_CERN_SCIENTIFIC_RESOURCE_TYPES + return resource_type_id in CERN_SCIENTIFIC_RESOURCE_TYPES def _build_communities(self): """Combine the configured target communities with the record's own.""" diff --git a/tests/cds-rdm/test_full_migration.py b/tests/cds-rdm/test_full_migration.py index 6e0e9cfd..a2249c0f 100644 --- a/tests/cds-rdm/test_full_migration.py +++ b/tests/cds-rdm/test_full_migration.py @@ -527,6 +527,8 @@ def test_full_migration_stream( if record["legacy_recid"] == "2783104": file_restricted(loaded_rec) parent_access_fields(loaded_rec) + # Skip scientific community inclusion check since it has restricted files + continue if record["legacy_recid"] == "2046076": irregular_exp_field(loaded_rec) if record["legacy_recid"] == "2041388": diff --git a/tests/cds-rdm/test_scientific_community.py b/tests/cds-rdm/test_scientific_community.py index e7b79ab4..f3f31e33 100644 --- a/tests/cds-rdm/test_scientific_community.py +++ b/tests/cds-rdm/test_scientific_community.py @@ -13,7 +13,7 @@ from cds_migrator_kit.errors import MissingConfiguration from cds_migrator_kit.rdm.records.transform.config import ( - CDS_CERN_SCIENTIFIC_RESOURCE_TYPES, + CERN_SCIENTIFIC_RESOURCE_TYPES, ) from cds_migrator_kit.rdm.records.transform.transform import CDSToRDMRecordTransform @@ -43,6 +43,12 @@ def _test_record( } +def _test_entry(files_restricted=False): + """Build a minimal raw dump entry for community tests.""" + status = "restricted" if files_restricted else "" + return {"files": [{"status": status}]} + + @pytest.fixture def transform(tmp_path, community): """Transform instance with a collection community configured.""" @@ -61,7 +67,12 @@ def test_adds_scientific_community_for_public_research_test_record( self, transform, community, cern_scientific_community ): """Public research records are included in the CERN Scientific community.""" - result = transform._communities_ids({}, _test_record()) + record = _test_record() + record["json"]["files"] = {"enabled": True} + result = transform._communities_ids( + _test_entry(), + record, + ) assert result == { "ids": [str(community.id), str(cern_scientific_community.id)], @@ -73,7 +84,7 @@ def test_keep_collection_community_as_default( ): """Collection community remains the default when CERN Scientific community is added.""" result = transform._communities_ids( - {}, + _test_entry(), _test_record(communities=["test-community"]), ) @@ -84,13 +95,13 @@ def test_keep_collection_community_as_default( str(cern_scientific_community.id), ] - @pytest.mark.parametrize("resource_type", CDS_CERN_SCIENTIFIC_RESOURCE_TYPES) + @pytest.mark.parametrize("resource_type", CERN_SCIENTIFIC_RESOURCE_TYPES) def test_research_resource_types( self, transform, cern_scientific_community, resource_type ): """All configured public research resource types trigger inclusion.""" result = transform._communities_ids( - {}, _test_record(resource_type=resource_type) + _test_entry(), _test_record(resource_type=resource_type) ) assert str(cern_scientific_community.id) in result["ids"] @@ -100,7 +111,22 @@ def test_skip_restricted_test_record( self, transform, community, cern_scientific_community ): """Restricted records are not included in the CERN Research community.""" - result = transform._communities_ids({}, _test_record(access="restricted")) + result = transform._communities_ids( + _test_entry(), _test_record(access="restricted") + ) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_restricted_files( + self, transform, community, cern_scientific_community + ): + """Records with restricted files are not included in the CERN Scientific community.""" + result = transform._communities_ids( + _test_entry(files_restricted=True), _test_record() + ) assert result == { "ids": [str(community.id)], @@ -111,7 +137,9 @@ def test_skip_non_research_resource_type( self, transform, community, cern_scientific_community ): """Non-research resource types are not included in the CERN Scientific community.""" - result = transform._communities_ids({}, _test_record(resource_type="other")) + result = transform._communities_ids( + _test_entry(), _test_record(resource_type="other") + ) assert result == { "ids": [str(community.id)], @@ -130,7 +158,7 @@ def test_skip_when_stream_is_restricted( migration_logger=MagicMock(), ) - result = transform._communities_ids({}, _test_record()) + result = transform._communities_ids(_test_entry(), _test_record()) assert result == { "ids": [str(community.id)], @@ -144,4 +172,4 @@ def test_raise_when_cern_scientific_community_not_configured( monkeypatch.setitem(test_app.config, "CDS_CERN_SCIENTIFIC_COMMUNITY_ID", None) with pytest.raises(MissingConfiguration): - transform._communities_ids({}, _test_record()) + transform._communities_ids(_test_entry(), _test_record()) From c15a7cc1460dd7d8421ed8a3be8172ec9826556a Mon Sep 17 00:00:00 2001 From: Saksham Date: Tue, 4 Aug 2026 15:55:53 +0200 Subject: [PATCH 3/4] add(config): Auto submit to cern research for conferencenote type --- cds_migrator_kit/rdm/records/transform/config.py | 1 + 1 file changed, 1 insertion(+) diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index fdbb52c2..636d0838 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -171,6 +171,7 @@ "publication-section", "publication-conferencepaper", "publication-conferenceproceeding", + "publication-conferencenote", "publication-journal", "publication-article", "publication-preprint", From 7fb15ab9393b3f28e81831b84298be47c803a0e6 Mon Sep 17 00:00:00 2001 From: Saksham Date: Tue, 6 Oct 2026 16:45:32 +0200 Subject: [PATCH 4/4] fix(tests): Update scientific community inclusion tests * Updated to be in sync with the recent codebase refactor --- .../rdm/records/transform/entities/parent.py | 16 +-- tests/cds-rdm/test_scientific_community.py | 126 +++++++++--------- 2 files changed, 69 insertions(+), 73 deletions(-) diff --git a/cds_migrator_kit/rdm/records/transform/entities/parent.py b/cds_migrator_kit/rdm/records/transform/entities/parent.py index 020227c4..c96a8bb1 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/parent.py +++ b/cds_migrator_kit/rdm/records/transform/entities/parent.py @@ -13,11 +13,14 @@ from invenio_accounts.models import User from sqlalchemy.exc import NoResultFound -from cds_migrator_kit.errors import ManualImportRequired, UnexpectedValue +from cds_migrator_kit.errors import ( + ManualImportRequired, + MissingConfiguration, + UnexpectedValue, +) +from cds_migrator_kit.rdm.records.transform.config import CERN_SCIENTIFIC_RESOURCE_TYPES from cds_migrator_kit.rdm.records.transform.mappers.base import RecordTransformContext from cds_migrator_kit.rdm.records.transform.mappers.record import AccessGrantsMapper -from cds_migrator_kit.rdm.records.transform.config import CERN_SCIENTIFIC_RESOURCE_TYPES -from cds_migrator_kit.errors import MissingConfiguration EMAIL_PATTERN = re.compile(r"[^@]+@[^@]+\.[^@]+") @@ -101,15 +104,12 @@ def _build_access(self): return {"owned_by": {"user": owner}} def _should_add_scientific_community(self): - if self.restricted or self.record.access_status != "public": + if self.record.restricted or self.record.access_status != "public": return False if any(file.get("status") for file in self.dojson_entry.get("files", [])): return False resource_type_id = ( - self.record.body - .get("metadata", {}) - .get("resource_type", {}) - .get("id") + self.record.body.get("metadata", {}).get("resource_type", {}).get("id") ) return resource_type_id in CERN_SCIENTIFIC_RESOURCE_TYPES diff --git a/tests/cds-rdm/test_scientific_community.py b/tests/cds-rdm/test_scientific_community.py index f3f31e33..752605e9 100644 --- a/tests/cds-rdm/test_scientific_community.py +++ b/tests/cds-rdm/test_scientific_community.py @@ -7,7 +7,7 @@ """Tests for auto-inclusion in the CERN Research community.""" -from unittest.mock import MagicMock +from types import SimpleNamespace import pytest @@ -15,16 +15,16 @@ from cds_migrator_kit.rdm.records.transform.config import ( CERN_SCIENTIFIC_RESOURCE_TYPES, ) -from cds_migrator_kit.rdm.records.transform.transform import CDSToRDMRecordTransform +from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent -def _test_record( +def _record( access="public", resource_type="publication-preprint", - communities=[], + restricted=False, recid="123456", ): - """Build a minimal CDSToRDMRecordEntry.transform() output for community tests.""" + """Build a minimal RecordEntry stand-in (only what RecordParent reads).""" metadata = { "title": "Test record", "publication_date": "2020-01-01", @@ -32,47 +32,46 @@ def _test_record( if resource_type is not None: metadata["resource_type"] = {"id": resource_type} - return { - "recid": recid, - "access": access, - "communities": communities, - "json": { - "files": {"enabled": False}, - "metadata": metadata, - }, - } + return SimpleNamespace( + recid=recid, + access_status=access, + restricted=restricted, + body={"metadata": metadata}, + ) -def _test_entry(files_restricted=False): - """Build a minimal raw dump entry for community tests.""" +def _dojson_entry(files_restricted=False, communities=None): + """Build a minimal DOJSON-processed entry for community tests.""" status = "restricted" if files_restricted else "" - return {"files": [{"status": status}]} + return {"files": [{"status": status}], "communities": communities or []} -@pytest.fixture -def transform(tmp_path, community): - """Transform instance with a collection community configured.""" - return CDSToRDMRecordTransform( - files_dump_dir=tmp_path, - missing_users=tmp_path, - communities_ids=[str(community.id)], - migration_logger=MagicMock(), +def _build_communities(communities_ids, record, dojson_entry): + """Run RecordParent's community resolution.""" + parent = RecordParent( + record=record, + raw_dump_entry={"recid": record.recid}, + dojson_entry=dojson_entry, + communities_ids=communities_ids, + access_grants_view=None, ) + return parent._build_communities() + + +@pytest.fixture +def communities_ids(community): + """Collection community configured for the migration run.""" + return [str(community.id)] -class TestCommunitiesIds: - """Test CDSToRDMRecordTransform._communities_ids().""" +class TestBuildCommunities: + """Test RecordParent._build_communities().""" - def test_adds_scientific_community_for_public_research_test_record( - self, transform, community, cern_scientific_community + def test_adds_scientific_community_for_public_research_record( + self, communities_ids, community, cern_scientific_community ): """Public research records are included in the CERN Scientific community.""" - record = _test_record() - record["json"]["files"] = {"enabled": True} - result = transform._communities_ids( - _test_entry(), - record, - ) + result = _build_communities(communities_ids, _record(), _dojson_entry()) assert result == { "ids": [str(community.id), str(cern_scientific_community.id)], @@ -80,12 +79,13 @@ def test_adds_scientific_community_for_public_research_test_record( } def test_keep_collection_community_as_default( - self, transform, community, cern_scientific_community + self, communities_ids, community, cern_scientific_community ): """Collection community remains the default when CERN Scientific community is added.""" - result = transform._communities_ids( - _test_entry(), - _test_record(communities=["test-community"]), + result = _build_communities( + communities_ids, + _record(), + _dojson_entry(communities=["test-community"]), ) assert result["default"] == str(community.id) @@ -97,22 +97,24 @@ def test_keep_collection_community_as_default( @pytest.mark.parametrize("resource_type", CERN_SCIENTIFIC_RESOURCE_TYPES) def test_research_resource_types( - self, transform, cern_scientific_community, resource_type + self, communities_ids, cern_scientific_community, resource_type ): """All configured public research resource types trigger inclusion.""" - result = transform._communities_ids( - _test_entry(), _test_record(resource_type=resource_type) + result = _build_communities( + communities_ids, + _record(resource_type=resource_type), + _dojson_entry(), ) assert str(cern_scientific_community.id) in result["ids"] - assert cern_scientific_community.id != result["default"] + assert str(cern_scientific_community.id) != result["default"] - def test_skip_restricted_test_record( - self, transform, community, cern_scientific_community + def test_skip_restricted_record( + self, communities_ids, community, cern_scientific_community ): """Restricted records are not included in the CERN Research community.""" - result = transform._communities_ids( - _test_entry(), _test_record(access="restricted") + result = _build_communities( + communities_ids, _record(access="restricted"), _dojson_entry() ) assert result == { @@ -121,11 +123,11 @@ def test_skip_restricted_test_record( } def test_skip_restricted_files( - self, transform, community, cern_scientific_community + self, communities_ids, community, cern_scientific_community ): """Records with restricted files are not included in the CERN Scientific community.""" - result = transform._communities_ids( - _test_entry(files_restricted=True), _test_record() + result = _build_communities( + communities_ids, _record(), _dojson_entry(files_restricted=True) ) assert result == { @@ -134,11 +136,11 @@ def test_skip_restricted_files( } def test_skip_non_research_resource_type( - self, transform, community, cern_scientific_community + self, communities_ids, community, cern_scientific_community ): """Non-research resource types are not included in the CERN Scientific community.""" - result = transform._communities_ids( - _test_entry(), _test_record(resource_type="other") + result = _build_communities( + communities_ids, _record(resource_type="other"), _dojson_entry() ) assert result == { @@ -147,29 +149,23 @@ def test_skip_non_research_resource_type( } def test_skip_when_stream_is_restricted( - self, tmp_path, community, cern_scientific_community + self, communities_ids, community, cern_scientific_community ): """Records on restricted migration streams are not included in the CERN Scientific community.""" - transform = CDSToRDMRecordTransform( - files_dump_dir=tmp_path, - missing_users=tmp_path, - communities_ids=[str(community.id)], - restricted=True, - migration_logger=MagicMock(), + result = _build_communities( + communities_ids, _record(restricted=True), _dojson_entry() ) - result = transform._communities_ids(_test_entry(), _test_record()) - assert result == { "ids": [str(community.id)], "default": str(community.id), } def test_raise_when_cern_scientific_community_not_configured( - self, test_app, transform, community, monkeypatch + self, test_app, communities_ids, monkeypatch ): - """No CERN Scientific community is added when config is unset.""" + """Missing CERN Scientific community config raises.""" monkeypatch.setitem(test_app.config, "CDS_CERN_SCIENTIFIC_COMMUNITY_ID", None) with pytest.raises(MissingConfiguration): - transform._communities_ids(_test_entry(), _test_record()) + _build_communities(communities_ids, _record(), _dojson_entry())