Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions cds_migrator_kit/errors.py
Original file line number Diff line number Diff line change
Expand Up @@ -85,3 +85,9 @@ class RecordFlaggedCuration(CDSMigrationException):
"""Record statistics error."""

description = "[Record needs to be curated]"


class MissingConfiguration(CDSMigrationException):
"""Missing configuration exception."""

description = "[Missing configuration]"
17 changes: 17 additions & 0 deletions cds_migrator_kit/rdm/records/transform/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -163,3 +163,20 @@
EXPERIMENT_ALIASES = {
"t2k": "re13",
}

# Public research publication resource types that are auto-included in the CERN Research community.
CERN_SCIENTIFIC_RESOURCE_TYPES = {
"publication-dissertation", # Already included by the migrator for thesis records.
"publication-book",
"publication-section",
"publication-conferencepaper",
"publication-conferenceproceeding",

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

to add: 'publication-conferencenote', added recently

"publication-conferencenote",
"publication-journal",
"publication-article",
"publication-preprint",
"publication-report",
"publication-technicalnote",
"publication-note",
"publication",
}
29 changes: 28 additions & 1 deletion cds_migrator_kit/rdm/records/transform/entities/parent.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,12 @@
from invenio_accounts.models import User
from sqlalchemy.exc import NoResultFound

from cds_migrator_kit.errors import ManualImportRequired, UnexpectedValue
from cds_migrator_kit.errors import (
ManualImportRequired,
MissingConfiguration,
UnexpectedValue,
)
from cds_migrator_kit.rdm.records.transform.config import CERN_SCIENTIFIC_RESOURCE_TYPES
from cds_migrator_kit.rdm.records.transform.mappers.base import RecordTransformContext
from cds_migrator_kit.rdm.records.transform.mappers.record import AccessGrantsMapper

Expand Down Expand Up @@ -98,10 +103,32 @@ def _build_access(self):
)
return {"owned_by": {"user": owner}}

def _should_add_scientific_community(self):
if self.record.restricted or self.record.access_status != "public":
return False
if any(file.get("status") for file in self.dojson_entry.get("files", [])):
return False
resource_type_id = (
self.record.body.get("metadata", {}).get("resource_type", {}).get("id")
)
return resource_type_id in CERN_SCIENTIFIC_RESOURCE_TYPES

def _build_communities(self):
"""Combine the configured target communities with the record's own."""
communities = self.dojson_entry.pop("communities", [])
communities = self.communities_ids + [slug for slug in communities]

scientific_community = current_app.config.get(
"CDS_CERN_SCIENTIFIC_COMMUNITY_ID"
)
if not scientific_community:
raise MissingConfiguration(
"CDS_CERN_SCIENTIFIC_COMMUNITY_ID is not configured"
)
if self._should_add_scientific_community():
if scientific_community not in communities:
communities.append(scientific_community)

if communities:
return {"ids": communities, "default": self.communities_ids[0]}
return {}
Expand Down
2 changes: 2 additions & 0 deletions cds_migrator_kit/rdm/records/transform/transform.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@

from cds_migrator_kit.errors import (
ManualImportRequired,
MissingConfiguration,
MissingRequiredField,
MultipleModelsMatched,
RestrictedFileDetected,
Expand Down Expand Up @@ -208,6 +209,7 @@ def _transform(self, raw_dump_entry) -> Optional[MigrationEntry]:
ManualImportRequired,
MissingRequiredField,
MultipleModelsMatched,
MissingConfiguration,
) as e:
migration_logger.add_log(e, record=raw_dump_entry)

Expand Down
14 changes: 13 additions & 1 deletion tests/cds-rdm/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -278,7 +278,7 @@ def running_app(


@pytest.fixture
def test_app(running_app):
def test_app(running_app, cern_scientific_community):
"""Get current app."""
running_app.app.config["RDM_PERSISTENT_IDENTIFIERS"]["doi"]["required"] = False
running_app.app.config["RDM_PARENT_PERSISTENT_IDENTIFIERS"]["doi"][
Expand Down Expand Up @@ -1723,6 +1723,18 @@ def community(running_app, db):
return comm


@pytest.fixture()
def cern_scientific_community(running_app, db, app):
"""A CERN Research community fixture."""
comm = Community.create({})
comm.slug = "cern-research"
comm.metadata = {"title": "CERN Research"}
comm.commit()
db.session.commit()
app.config["CDS_CERN_SCIENTIFIC_COMMUNITY_ID"] = str(comm.id)
return comm


# @pytest.fixture()
# def users(app, db):
# """Create example user."""
Expand Down
5 changes: 0 additions & 5 deletions tests/cds-rdm/test_bulletin_issue.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,8 +80,3 @@ def test_bulletin_issue(
)
if record["legacy_recid"] == "2234683":
parent_related_identifier(loaded_rec)


# FAILED tests/cds-rdm/test_access_permissions.py::test_access_permissions - assert 2 == 3
# FAILED tests/cds-rdm/test_bulletin_issue.py::test_bulletin_issue - AssertionError: assert [{'identifier...heme': 'cds'}] == [{'identifier...heme': 'cds'}]
# FAILED tests/cds-rdm/test_full_migration.py::test_full_migration_stream - assert 0 == 2
7 changes: 4 additions & 3 deletions tests/cds-rdm/test_ep_approval_entry.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@
EPPHAPP_FILE_TYPE,
PublicEntry,
RestrictedEntry,
_cern_scientific_community_id,
)
from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent
from cds_migrator_kit.rdm.records.transform.entities.record import RecordEntry
Expand Down Expand Up @@ -568,20 +569,20 @@ def test_public_adds_cern_scientific_community(self, app):
entry, _make_approval_request(), _make_migration_logger()
).build()

assert CDS_CERN_SCIENTIFIC_COMMUNITY_ID in result["parent"].communities["ids"]
assert _cern_scientific_community_id() in result["parent"].communities["ids"]

def test_public_does_not_duplicate_community(self, app):
entry = _make_entry(_versions_with_epphapp())
entry["parent"].communities["ids"] = [
"example-community",
CDS_CERN_SCIENTIFIC_COMMUNITY_ID,
_cern_scientific_community_id(),
]
result = PublicEntry(
entry, _make_approval_request(), _make_migration_logger()
).build()

community_ids = result["parent"].communities["ids"]
assert community_ids.count(CDS_CERN_SCIENTIFIC_COMMUNITY_ID) == 1
assert community_ids.count(_cern_scientific_community_id()) == 1


class TestEntryImmutability:
Expand Down
11 changes: 11 additions & 0 deletions tests/cds-rdm/test_full_migration.py
Original file line number Diff line number Diff line change
Expand Up @@ -482,6 +482,7 @@ def test_full_migration_stream(
superuser_identity,
orcid_name_data,
community,
cern_scientific_community, # Fixture for the CERN Scientific community, so that it gets auto included for PUBLIC-ations.
mocker,
groups,
):
Expand Down Expand Up @@ -526,6 +527,8 @@ def test_full_migration_stream(
if record["legacy_recid"] == "2783104":
file_restricted(loaded_rec)
parent_access_fields(loaded_rec)
# Skip scientific community inclusion check since it has restricted files
continue
if record["legacy_recid"] == "2046076":
irregular_exp_field(loaded_rec)
if record["legacy_recid"] == "2041388":
Expand All @@ -537,6 +540,8 @@ def test_full_migration_stream(
if record["legacy_recid"] == "2294138":
author_with_inspire(loaded_rec)

# Check if the record is also included in the CERN Scientific community since it is a publication report
scientific_community_inclusion(loaded_rec, cern_scientific_community.id)
# Check if remote account has the correct metadata
# Check if user profile has the correct metadata
user_metadata()
Expand Down Expand Up @@ -606,3 +611,9 @@ def user_metadata():
"person_id": "11115",
"department": "IT",
}


def scientific_community_inclusion(record, cern_scientific_community_uuid):
"""Checks if the record is included in the CERN Scientific community."""
assert str(cern_scientific_community_uuid) in record._record.parent.communities.ids
assert cern_scientific_community_uuid != record._record.parent.communities.default
171 changes: 171 additions & 0 deletions tests/cds-rdm/test_scientific_community.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,171 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2026 CERN.
#
# CDS-RDM is free software; you can redistribute it and/or modify it under
# the terms of the MIT License; see LICENSE file for more details.

"""Tests for auto-inclusion in the CERN Research community."""

from types import SimpleNamespace

import pytest

from cds_migrator_kit.errors import MissingConfiguration
from cds_migrator_kit.rdm.records.transform.config import (
CERN_SCIENTIFIC_RESOURCE_TYPES,
)
from cds_migrator_kit.rdm.records.transform.entities.parent import RecordParent


def _record(
access="public",
resource_type="publication-preprint",
restricted=False,
recid="123456",
):
"""Build a minimal RecordEntry stand-in (only what RecordParent reads)."""
metadata = {
"title": "Test record",
"publication_date": "2020-01-01",
}
if resource_type is not None:
metadata["resource_type"] = {"id": resource_type}

return SimpleNamespace(
recid=recid,
access_status=access,
restricted=restricted,
body={"metadata": metadata},
)


def _dojson_entry(files_restricted=False, communities=None):
"""Build a minimal DOJSON-processed entry for community tests."""
status = "restricted" if files_restricted else ""
return {"files": [{"status": status}], "communities": communities or []}


def _build_communities(communities_ids, record, dojson_entry):
"""Run RecordParent's community resolution."""
parent = RecordParent(
record=record,
raw_dump_entry={"recid": record.recid},
dojson_entry=dojson_entry,
communities_ids=communities_ids,
access_grants_view=None,
)
return parent._build_communities()


@pytest.fixture
def communities_ids(community):
"""Collection community configured for the migration run."""
return [str(community.id)]


class TestBuildCommunities:
"""Test RecordParent._build_communities()."""

def test_adds_scientific_community_for_public_research_record(
self, communities_ids, community, cern_scientific_community
):
"""Public research records are included in the CERN Scientific community."""
result = _build_communities(communities_ids, _record(), _dojson_entry())

assert result == {
"ids": [str(community.id), str(cern_scientific_community.id)],
"default": str(community.id),
}

def test_keep_collection_community_as_default(
self, communities_ids, community, cern_scientific_community
):
"""Collection community remains the default when CERN Scientific community is added."""
result = _build_communities(
communities_ids,
_record(),
_dojson_entry(communities=["test-community"]),
)

assert result["default"] == str(community.id)
assert result["ids"] == [
str(community.id),
"test-community",
str(cern_scientific_community.id),
]

@pytest.mark.parametrize("resource_type", CERN_SCIENTIFIC_RESOURCE_TYPES)
def test_research_resource_types(
self, communities_ids, cern_scientific_community, resource_type
):
"""All configured public research resource types trigger inclusion."""
result = _build_communities(
communities_ids,
_record(resource_type=resource_type),
_dojson_entry(),
)

assert str(cern_scientific_community.id) in result["ids"]
assert str(cern_scientific_community.id) != result["default"]

def test_skip_restricted_record(
self, communities_ids, community, cern_scientific_community
):
"""Restricted records are not included in the CERN Research community."""
result = _build_communities(
communities_ids, _record(access="restricted"), _dojson_entry()
)

assert result == {
"ids": [str(community.id)],
"default": str(community.id),
}

def test_skip_restricted_files(
self, communities_ids, community, cern_scientific_community
):
"""Records with restricted files are not included in the CERN Scientific community."""
result = _build_communities(
communities_ids, _record(), _dojson_entry(files_restricted=True)
)

assert result == {
"ids": [str(community.id)],
"default": str(community.id),
}

def test_skip_non_research_resource_type(
self, communities_ids, community, cern_scientific_community
):
"""Non-research resource types are not included in the CERN Scientific community."""
result = _build_communities(
communities_ids, _record(resource_type="other"), _dojson_entry()
)

assert result == {
"ids": [str(community.id)],
"default": str(community.id),
}

def test_skip_when_stream_is_restricted(
self, communities_ids, community, cern_scientific_community
):
"""Records on restricted migration streams are not included in the CERN Scientific community."""
result = _build_communities(
communities_ids, _record(restricted=True), _dojson_entry()
)

assert result == {
"ids": [str(community.id)],
"default": str(community.id),
}

def test_raise_when_cern_scientific_community_not_configured(
self, test_app, communities_ids, monkeypatch
):
"""Missing CERN Scientific community config raises."""
monkeypatch.setitem(test_app.config, "CDS_CERN_SCIENTIFIC_COMMUNITY_ID", None)

with pytest.raises(MissingConfiguration):
_build_communities(communities_ids, _record(), _dojson_entry())
Loading