diff --git a/cds_migrator_kit/errors.py b/cds_migrator_kit/errors.py index 5cbe912d..a998ac4a 100644 --- a/cds_migrator_kit/errors.py +++ b/cds_migrator_kit/errors.py @@ -85,3 +85,9 @@ class RecordFlaggedCuration(CDSMigrationException): """Record statistics error.""" description = "[Record needs to be curated]" + + +class MissingConfiguration(CDSMigrationException): + """Missing configuration exception.""" + + description = "[Missing configuration]" diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 2bca4a47..636d0838 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -163,3 +163,20 @@ EXPERIMENT_ALIASES = { "t2k": "re13", } + +# Public research publication resource types that are auto-included in the CERN Research community. +CERN_SCIENTIFIC_RESOURCE_TYPES = { + "publication-dissertation", # Already included by the migrator for thesis records. + "publication-book", + "publication-section", + "publication-conferencepaper", + "publication-conferenceproceeding", + "publication-conferencenote", + "publication-journal", + "publication-article", + "publication-preprint", + "publication-report", + "publication-technicalnote", + "publication-note", + "publication", +} diff --git a/cds_migrator_kit/rdm/records/transform/entities/parent.py b/cds_migrator_kit/rdm/records/transform/entities/parent.py index adfd130a..c96a8bb1 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/parent.py +++ b/cds_migrator_kit/rdm/records/transform/entities/parent.py @@ -13,7 +13,12 @@ from invenio_accounts.models import User from sqlalchemy.exc import NoResultFound -from cds_migrator_kit.errors import ManualImportRequired, UnexpectedValue +from cds_migrator_kit.errors import ( + ManualImportRequired, + MissingConfiguration, + UnexpectedValue, +) +from cds_migrator_kit.rdm.records.transform.config import CERN_SCIENTIFIC_RESOURCE_TYPES from cds_migrator_kit.rdm.records.transform.mappers.base import RecordTransformContext from cds_migrator_kit.rdm.records.transform.mappers.record import AccessGrantsMapper @@ -98,10 +103,32 @@ def _build_access(self): ) return {"owned_by": {"user": owner}} + def _should_add_scientific_community(self): + if self.record.restricted or self.record.access_status != "public": + return False + if any(file.get("status") for file in self.dojson_entry.get("files", [])): + return False + resource_type_id = ( + self.record.body.get("metadata", {}).get("resource_type", {}).get("id") + ) + return resource_type_id in CERN_SCIENTIFIC_RESOURCE_TYPES + def _build_communities(self): """Combine the configured target communities with the record's own.""" communities = self.dojson_entry.pop("communities", []) communities = self.communities_ids + [slug for slug in communities] + + scientific_community = current_app.config.get( + "CDS_CERN_SCIENTIFIC_COMMUNITY_ID" + ) + if not scientific_community: + raise MissingConfiguration( + "CDS_CERN_SCIENTIFIC_COMMUNITY_ID is not configured" + ) + if self._should_add_scientific_community(): + if scientific_community not in communities: + communities.append(scientific_community) + if communities: return {"ids": communities, "default": self.communities_ids[0]} return {} diff --git a/cds_migrator_kit/rdm/records/transform/transform.py b/cds_migrator_kit/rdm/records/transform/transform.py index bcb60b1b..8c362175 100644 --- a/cds_migrator_kit/rdm/records/transform/transform.py +++ b/cds_migrator_kit/rdm/records/transform/transform.py @@ -27,6 +27,7 @@ from cds_migrator_kit.errors import ( ManualImportRequired, + MissingConfiguration, MissingRequiredField, MultipleModelsMatched, RestrictedFileDetected, @@ -208,6 +209,7 @@ def _transform(self, raw_dump_entry) -> Optional[MigrationEntry]: ManualImportRequired, MissingRequiredField, MultipleModelsMatched, + MissingConfiguration, ) as e: migration_logger.add_log(e, record=raw_dump_entry) diff --git a/tests/cds-rdm/conftest.py b/tests/cds-rdm/conftest.py index 8cf329e7..41bbead0 100644 --- a/tests/cds-rdm/conftest.py +++ b/tests/cds-rdm/conftest.py @@ -278,7 +278,7 @@ def running_app( @pytest.fixture -def test_app(running_app): +def test_app(running_app, cern_scientific_community): """Get current app.""" running_app.app.config["RDM_PERSISTENT_IDENTIFIERS"]["doi"]["required"] = False running_app.app.config["RDM_PARENT_PERSISTENT_IDENTIFIERS"]["doi"][ @@ -1723,6 +1723,18 @@ def community(running_app, db): return comm +@pytest.fixture() +def cern_scientific_community(running_app, db, app): + """A CERN Research community fixture.""" + comm = Community.create({}) + comm.slug = "cern-research" + comm.metadata = {"title": "CERN Research"} + comm.commit() + db.session.commit() + app.config["CDS_CERN_SCIENTIFIC_COMMUNITY_ID"] = str(comm.id) + return comm + + # @pytest.fixture() # def users(app, db): # """Create example user.""" diff --git a/tests/cds-rdm/test_bulletin_issue.py b/tests/cds-rdm/test_bulletin_issue.py index 88cd0bdf..12c2d0f9 100644 --- a/tests/cds-rdm/test_bulletin_issue.py +++ b/tests/cds-rdm/test_bulletin_issue.py @@ -80,8 +80,3 @@ def test_bulletin_issue( ) if record["legacy_recid"] == "2234683": parent_related_identifier(loaded_rec) - - -# FAILED tests/cds-rdm/test_access_permissions.py::test_access_permissions - assert 2 == 3 -# FAILED tests/cds-rdm/test_bulletin_issue.py::test_bulletin_issue - AssertionError: assert [{'identifier...heme': 'cds'}] == [{'identifier...heme': 'cds'}] -# FAILED tests/cds-rdm/test_full_migration.py::test_full_migration_stream - assert 0 == 2 diff --git a/tests/cds-rdm/test_ep_approval_entry.py b/tests/cds-rdm/test_ep_approval_entry.py index 3cb97593..d028edd9 100644 --- a/tests/cds-rdm/test_ep_approval_entry.py +++ b/tests/cds-rdm/test_ep_approval_entry.py @@ -19,6 +19,7 @@ EPPHAPP_FILE_TYPE, PublicEntry, RestrictedEntry, + _cern_scientific_community_id, ) from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent from cds_migrator_kit.rdm.records.transform.entities.record import RecordEntry @@ -568,20 +569,20 @@ def test_public_adds_cern_scientific_community(self, app): entry, _make_approval_request(), _make_migration_logger() ).build() - assert CDS_CERN_SCIENTIFIC_COMMUNITY_ID in result["parent"].communities["ids"] + assert _cern_scientific_community_id() in result["parent"].communities["ids"] def test_public_does_not_duplicate_community(self, app): entry = _make_entry(_versions_with_epphapp()) entry["parent"].communities["ids"] = [ "example-community", - CDS_CERN_SCIENTIFIC_COMMUNITY_ID, + _cern_scientific_community_id(), ] result = PublicEntry( entry, _make_approval_request(), _make_migration_logger() ).build() community_ids = result["parent"].communities["ids"] - assert community_ids.count(CDS_CERN_SCIENTIFIC_COMMUNITY_ID) == 1 + assert community_ids.count(_cern_scientific_community_id()) == 1 class TestEntryImmutability: diff --git a/tests/cds-rdm/test_full_migration.py b/tests/cds-rdm/test_full_migration.py index ee45fa23..a2249c0f 100644 --- a/tests/cds-rdm/test_full_migration.py +++ b/tests/cds-rdm/test_full_migration.py @@ -482,6 +482,7 @@ def test_full_migration_stream( superuser_identity, orcid_name_data, community, + cern_scientific_community, # Fixture for the CERN Scientific community, so that it gets auto included for PUBLIC-ations. mocker, groups, ): @@ -526,6 +527,8 @@ def test_full_migration_stream( if record["legacy_recid"] == "2783104": file_restricted(loaded_rec) parent_access_fields(loaded_rec) + # Skip scientific community inclusion check since it has restricted files + continue if record["legacy_recid"] == "2046076": irregular_exp_field(loaded_rec) if record["legacy_recid"] == "2041388": @@ -537,6 +540,8 @@ def test_full_migration_stream( if record["legacy_recid"] == "2294138": author_with_inspire(loaded_rec) + # Check if the record is also included in the CERN Scientific community since it is a publication report + scientific_community_inclusion(loaded_rec, cern_scientific_community.id) # Check if remote account has the correct metadata # Check if user profile has the correct metadata user_metadata() @@ -606,3 +611,9 @@ def user_metadata(): "person_id": "11115", "department": "IT", } + + +def scientific_community_inclusion(record, cern_scientific_community_uuid): + """Checks if the record is included in the CERN Scientific community.""" + assert str(cern_scientific_community_uuid) in record._record.parent.communities.ids + assert cern_scientific_community_uuid != record._record.parent.communities.default diff --git a/tests/cds-rdm/test_scientific_community.py b/tests/cds-rdm/test_scientific_community.py new file mode 100644 index 00000000..752605e9 --- /dev/null +++ b/tests/cds-rdm/test_scientific_community.py @@ -0,0 +1,171 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for auto-inclusion in the CERN Research community.""" + +from types import SimpleNamespace + +import pytest + +from cds_migrator_kit.errors import MissingConfiguration +from cds_migrator_kit.rdm.records.transform.config import ( + CERN_SCIENTIFIC_RESOURCE_TYPES, +) +from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent + + +def _record( + access="public", + resource_type="publication-preprint", + restricted=False, + recid="123456", +): + """Build a minimal RecordEntry stand-in (only what RecordParent reads).""" + metadata = { + "title": "Test record", + "publication_date": "2020-01-01", + } + if resource_type is not None: + metadata["resource_type"] = {"id": resource_type} + + return SimpleNamespace( + recid=recid, + access_status=access, + restricted=restricted, + body={"metadata": metadata}, + ) + + +def _dojson_entry(files_restricted=False, communities=None): + """Build a minimal DOJSON-processed entry for community tests.""" + status = "restricted" if files_restricted else "" + return {"files": [{"status": status}], "communities": communities or []} + + +def _build_communities(communities_ids, record, dojson_entry): + """Run RecordParent's community resolution.""" + parent = RecordParent( + record=record, + raw_dump_entry={"recid": record.recid}, + dojson_entry=dojson_entry, + communities_ids=communities_ids, + access_grants_view=None, + ) + return parent._build_communities() + + +@pytest.fixture +def communities_ids(community): + """Collection community configured for the migration run.""" + return [str(community.id)] + + +class TestBuildCommunities: + """Test RecordParent._build_communities().""" + + def test_adds_scientific_community_for_public_research_record( + self, communities_ids, community, cern_scientific_community + ): + """Public research records are included in the CERN Scientific community.""" + result = _build_communities(communities_ids, _record(), _dojson_entry()) + + assert result == { + "ids": [str(community.id), str(cern_scientific_community.id)], + "default": str(community.id), + } + + def test_keep_collection_community_as_default( + self, communities_ids, community, cern_scientific_community + ): + """Collection community remains the default when CERN Scientific community is added.""" + result = _build_communities( + communities_ids, + _record(), + _dojson_entry(communities=["test-community"]), + ) + + assert result["default"] == str(community.id) + assert result["ids"] == [ + str(community.id), + "test-community", + str(cern_scientific_community.id), + ] + + @pytest.mark.parametrize("resource_type", CERN_SCIENTIFIC_RESOURCE_TYPES) + def test_research_resource_types( + self, communities_ids, cern_scientific_community, resource_type + ): + """All configured public research resource types trigger inclusion.""" + result = _build_communities( + communities_ids, + _record(resource_type=resource_type), + _dojson_entry(), + ) + + assert str(cern_scientific_community.id) in result["ids"] + assert str(cern_scientific_community.id) != result["default"] + + def test_skip_restricted_record( + self, communities_ids, community, cern_scientific_community + ): + """Restricted records are not included in the CERN Research community.""" + result = _build_communities( + communities_ids, _record(access="restricted"), _dojson_entry() + ) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_restricted_files( + self, communities_ids, community, cern_scientific_community + ): + """Records with restricted files are not included in the CERN Scientific community.""" + result = _build_communities( + communities_ids, _record(), _dojson_entry(files_restricted=True) + ) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_non_research_resource_type( + self, communities_ids, community, cern_scientific_community + ): + """Non-research resource types are not included in the CERN Scientific community.""" + result = _build_communities( + communities_ids, _record(resource_type="other"), _dojson_entry() + ) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_skip_when_stream_is_restricted( + self, communities_ids, community, cern_scientific_community + ): + """Records on restricted migration streams are not included in the CERN Scientific community.""" + result = _build_communities( + communities_ids, _record(restricted=True), _dojson_entry() + ) + + assert result == { + "ids": [str(community.id)], + "default": str(community.id), + } + + def test_raise_when_cern_scientific_community_not_configured( + self, test_app, communities_ids, monkeypatch + ): + """Missing CERN Scientific community config raises.""" + monkeypatch.setitem(test_app.config, "CDS_CERN_SCIENTIFIC_COMMUNITY_ID", None) + + with pytest.raises(MissingConfiguration): + _build_communities(communities_ids, _record(), _dojson_entry())