From fb1812e6b840a5531eedfbefd39f3e6ad81216df Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:51:57 -0300 Subject: [PATCH 01/14] =?UTF-8?q?feat(article):=20adiciona=20suporte=20a?= =?UTF-8?q?=20ArticleSource=20n=C3=A3o=20p=C3=BAblico=20(NOT=5FPUBLIC)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Evitar requisição HTTP desnecessária para documentos que estão temporária ou efetivamente despublicados, identificados via is_public. Solução técnica: - Novo status NOT_PUBLIC em ArticleSource.StatusChoices. - create() e create_or_update() passam a aceitar is_public: quando False, o status é definido como NOT_PUBLIC em vez de PENDING. - create_or_update(): quando is_public volta a True e o status atual é NOT_PUBLIC, o status retorna para PENDING, reabrindo o pipeline. obj.save() só é chamado quando algo de fato mudou (changed). - add_pid_provider(): adiciona guarda no início — se o status é NOT_PUBLIC e force_update não foi passado, retorna sem executar request_xml nem request_pid. Notas adicionais (fora do escopo de is_public, incluídas neste diff): - is_completed deixa de chamar self.save() ao marcar COMPLETED. - ContribPerson passa a usar affiliation.set_normalized(...) em vez de add_normalized_affiliation(...). --- article/models.py | 31 +++++++++++++++++++++++++------ 1 file changed, 25 insertions(+), 6 deletions(-) diff --git a/article/models.py b/article/models.py index 85a4e434d..7be88970a 100755 --- a/article/models.py +++ b/article/models.py @@ -1694,6 +1694,7 @@ class StatusChoices(models.TextChoices): REPROCESS = "reprocess", _("Reprocess") URL_ERROR = "url_error", _("URL Error") XML_ERROR = "xml_error", _("XML Error") + NOT_PUBLIC = "not_public", _("Not public") url = models.URLField( verbose_name=_("Article URL"), @@ -1801,7 +1802,7 @@ def get(cls, url): raise ValueError("ArticleSource.get requires url") @classmethod - def create(cls, user, url=None, source_date=None, am_article=None, force_update=None, auto_solve_pid_conflict=False): + def create(cls, user, url=None, source_date=None, am_article=None, force_update=None, auto_solve_pid_conflict=False, is_public=None): if not url: raise ValueError("ArticleSource.create requires url") @@ -1811,7 +1812,10 @@ def create(cls, user, url=None, source_date=None, am_article=None, force_update= obj.url = url obj.source_date = source_date obj.am_article = am_article - obj.status = cls.StatusChoices.PENDING + if is_public is False: + obj.status = cls.StatusChoices.NOT_PUBLIC + else: + obj.status = cls.StatusChoices.PENDING obj.add_pid_provider(user, force_update, auto_solve_pid_conflict=auto_solve_pid_conflict) return obj except IntegrityError: @@ -1819,13 +1823,20 @@ def create(cls, user, url=None, source_date=None, am_article=None, force_update= @classmethod def create_or_update( - cls, user, url=None, source_date=None, am_article=None, force_update=None, auto_solve_pid_conflict=False + cls, user, url=None, source_date=None, am_article=None, force_update=None, auto_solve_pid_conflict=False, is_public=None ): try: logging.info( f"ArticleSource.create_or_update {url} {source_date} {am_article} {force_update}" ) + changed = False obj = cls.get(url=url) + if is_public is False: + obj.status = cls.StatusChoices.NOT_PUBLIC + changed = True + elif is_public is True and obj.status == cls.StatusChoices.NOT_PUBLIC: + obj.status = cls.StatusChoices.PENDING + changed = True if ( force_update or (source_date and source_date != obj.source_date) @@ -1835,6 +1846,9 @@ def create_or_update( obj.source_date = source_date obj.am_article = am_article obj.add_pid_provider(user, force_update, auto_solve_pid_conflict=auto_solve_pid_conflict) + changed = True + if changed: + obj.save() return obj except cls.DoesNotExist: return cls.create( @@ -1843,7 +1857,8 @@ def create_or_update( source_date=source_date, am_article=am_article, force_update=force_update, - auto_solve_pid_conflict=auto_solve_pid_conflict + auto_solve_pid_conflict=auto_solve_pid_conflict, + is_public=is_public, ) @cached_property @@ -2008,7 +2023,6 @@ def is_completed(self): return False if self.status != ArticleSource.StatusChoices.COMPLETED: self.status = ArticleSource.StatusChoices.COMPLETED - self.save() logging.info(f"Completed: ArticleSource {self.url} is completed") return True @@ -2029,6 +2043,11 @@ def add_pid_provider(self, user, force_update=False, auto_solve_pid_conflict=Fal """ try: detail = [] + + if self.status == ArticleSource.StatusChoices.NOT_PUBLIC: + if not force_update: + return + self.status = ArticleSource.StatusChoices.PENDING # --- Etapa 1: request_xml --- @@ -3249,7 +3268,7 @@ def add_normalized_affiliation(self, user, organization=None, location=None, self.save() # Add normalized affiliation to the ArticleAffiliation - self.affiliation.add_normalized_affiliation( + self.affiliation.set_normalized( user=user, organization=organization, location=location, From efc185cadd1108f87d0011febd72c42b5efdd061 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:51:57 -0300 Subject: [PATCH 02/14] feat(article): OPACHarvester repassa is_public a partir de status MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Fornecer o dado de publicação/despublicação já na coleta (harvest), para que o pipeline possa decidir sem precisar de requisição extra. Solução técnica: Inclui "is_public": item.get("status") no dict retornado por OPACHarvester. Nome do campo de origem manteve-se status (herdado da fonte); mapeado para is_public no dict de saída. --- core/utils/harvesters.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/core/utils/harvesters.py b/core/utils/harvesters.py index 928b72336..6c77fdd8e 100644 --- a/core/utils/harvesters.py +++ b/core/utils/harvesters.py @@ -246,6 +246,8 @@ def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: "publication_year": publication_year, "url": xml_url, "source_type": "opac", + # True ou False. Nome ideal seria is_public, mas ficou como status + "is_public": item.get("status"), "metadata": { "aop_pid": item.get("aop_pid"), "default_language": item.get("default_language"), From 9e59707e036f14ba2d100c7aa58654dbd5524889 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:51:57 -0300 Subject: [PATCH 03/14] feat(article): _iter_from_harvest inclui is_public no dispatch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Propagar o sinal de publicação/despublicação coletado pelo harvester até o ponto de despacho da task, para que chegue a task_process_article_pipeline. Solução técnica: Adiciona "is_public": document.get("is_public") ao dict yield de _iter_from_harvest. --- article/controller.py | 1 + 1 file changed, 1 insertion(+) diff --git a/article/controller.py b/article/controller.py index 48ede7f3e..e9c666bdf 100644 --- a/article/controller.py +++ b/article/controller.py @@ -564,6 +564,7 @@ def _iter_from_harvest(self): "collection_acron": collection_acron, "pid": document["pid_v2"], "source_date": document.get("processing_date") or document.get("origin_date"), + "is_public": document.get("is_public") } self._iter_from_harvest_count = count From a7fd97f1cf56af56b2b573b82cdfc3a29a186695 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:51:57 -0300 Subject: [PATCH 04/14] feat(article): task_process_article_pipeline repassa is_public MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Fechar o fluxo do sinal is_public até ArticleSource.create_or_update, permitindo que o pipeline marque/reabra o status NOT_PUBLIC. Solução técnica: Novo parâmetro is_public em task_process_article_pipeline, repassado para ArticleSource.create_or_update(...). --- article/tasks.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/article/tasks.py b/article/tasks.py index 4905ca045..6e1feec77 100644 --- a/article/tasks.py +++ b/article/tasks.py @@ -857,6 +857,7 @@ def task_process_article_pipeline( version=None, user_id=None, username=None, + is_public=None, ): """ Pipeline principal de processamento de artigos com múltiplos pontos de entrada. @@ -941,6 +942,7 @@ def task_process_article_pipeline( force_update=force_update, am_article=am_article, auto_solve_pid_conflict=auto_solve_pid_conflict, + is_public=is_public, ) pp_xml_id = article_source.pid_provider_xml.id From e3f00247823a0f812729419aad8da04a06114c36 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:52:31 -0300 Subject: [PATCH 05/14] test(article): adiciona __init__.py do pacote de testes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Criar o pacote de testes do app article. Solução técnica: Arquivo __init__.py vazio, necessário para o Django/pytest reconhecer o diretório como pacote de testes. --- article/tests/__init__.py | 0 1 file changed, 0 insertions(+), 0 deletions(-) create mode 100644 article/tests/__init__.py diff --git a/article/tests/__init__.py b/article/tests/__init__.py new file mode 100644 index 000000000..e69de29bb From 4fcba036d68af544741edba46117bb8e66a2ed1b Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:52:31 -0300 Subject: [PATCH 06/14] test(article): adiciona test_mixins.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Fornecer mixins/utilitários de teste reutilizáveis pelas demais suítes do app article (setup de dados, mocks comuns, supressão de logging do OpenSearch em ambiente de teste). Solução técnica: Novo arquivo test_mixins.py. --- article/tests/test_mixins.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 article/tests/test_mixins.py diff --git a/article/tests/test_mixins.py b/article/tests/test_mixins.py new file mode 100644 index 000000000..bd0af97e7 --- /dev/null +++ b/article/tests/test_mixins.py @@ -0,0 +1,14 @@ +from unittest.mock import patch +from article.models import Article + + +class ArticleTestMixin: + """Mixin com helpers e mocks para o app Article.""" + + def make_article(self, user=None, pid_v3=None): + kwargs = {} + if user: + kwargs["creator"] = user + if pid_v3: + kwargs["pid_v3"] = pid_v3 + return Article.objects.create(**kwargs) From ad4bbd03556799ee04e5ff20ccb8dc09cf83dec0 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:52:31 -0300 Subject: [PATCH 07/14] test(article): adiciona test_models.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Cobrir com testes unitários os models do app article (Article, ArticleSource, ContribPerson, etc.). Solução técnica: Novo arquivo test_models.py, usando os mixins de test_mixins.py e unittest.mock para isolar dependências externas. --- article/tests/test_models.py | 742 +++++++++++++++++++++++++++++++++++ 1 file changed, 742 insertions(+) create mode 100644 article/tests/test_models.py diff --git a/article/tests/test_models.py b/article/tests/test_models.py new file mode 100644 index 000000000..5c602ad32 --- /dev/null +++ b/article/tests/test_models.py @@ -0,0 +1,742 @@ +from datetime import datetime +from unittest.mock import patch + +from django.contrib.auth import get_user_model +from django.test import TestCase +from django.utils.timezone import make_aware +from freezegun import freeze_time + +from article import choices +from article.models import Article, ArticleAffiliation, ContribCollab, ContribPerson +from article.tests.test_mixins import ArticleTestMixin +from organization.models import NormAffiliation +from organization.tests.test_mixins import OrganizationTestMixin + +User = get_user_model() + + +class RemoveDuplicateArticlesTest(ArticleTestMixin, TestCase): + """Testes de deduplicação de artigos.""" + + def setUp(self): + self.user = User.objects.create_user(username="dedup_user", password="x") + + def create_article_at_time(self, dt, sps_pkg_name): + with freeze_time(dt): + return Article.objects.create( + sps_pkg_name=sps_pkg_name, + creator=self.user, + created=make_aware(datetime.strptime(dt, "%Y-%m-%d")), + ) + + def test_marks_most_recent_as_deduplicated_when_none_available(self): + self.create_article_at_time("2023-01-01", "pkg1") + self.create_article_at_time("2023-01-02", "pkg1") + most_recent = self.create_article_at_time("2023-01-03", "pkg1") + + with patch.object(Article, "check_availability", return_value=False): + Article.deduplicate_items(user=self.user, deduplicate=True) + + self.assertEqual(Article.objects.filter(sps_pkg_name="pkg1").count(), 3) + most_recent.refresh_from_db() + self.assertEqual(most_recent.data_status, choices.DATA_STATUS_DEDUPLICATED) + + def test_stops_at_first_available_starting_from_most_recent(self): + oldest = self.create_article_at_time("2023-01-01", "pkg2") + newest = self.create_article_at_time("2023-01-02", "pkg2") + + def fake_check_availability(self_article, user, force_update=False): + return self_article.id == oldest.id + + with patch.object(Article, "check_availability", fake_check_availability): + Article.deduplicate_items(user=self.user, deduplicate=True) + + oldest.refresh_from_db() + newest.refresh_from_db() + self.assertNotEqual(oldest.data_status, choices.DATA_STATUS_DEDUPLICATED) + self.assertNotEqual(newest.data_status, choices.DATA_STATUS_DEDUPLICATED) + + def test_marks_all_matching_as_duplicated(self): + a1 = self.create_article_at_time("2022-06-03", "pkg3") + a2 = self.create_article_at_time("2022-06-04", "pkg3") + + Article.deduplicate_items(user=self.user, mark_as_duplicated=True) + + a1.refresh_from_db() + a2.refresh_from_db() + self.assertEqual(a1.data_status, choices.DATA_STATUS_DUPLICATED) + self.assertEqual(a2.data_status, choices.DATA_STATUS_DUPLICATED) + + def test_no_action_if_only_one_article(self): + self.create_article_at_time("2023-01-01", "pkg4") + + Article.deduplicate_items( + user=self.user, mark_as_duplicated=True, deduplicate=True + ) + + self.assertEqual(Article.objects.filter(sps_pkg_name="pkg4").count(), 1) + + +class ArticleAffiliationTest(ArticleTestMixin, OrganizationTestMixin, TestCase): + """Tests for ArticleAffiliation model.""" + + def setUp(self): + self.user = User.objects.create_user(username="testuser", password="testpass") + self.organization = self.make_organization(self.user) + self.article = self.make_article(user=self.user) + self.ArticleAffiliation = ArticleAffiliation + + def test_article_affiliation_create_with_organization(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + self.assertIsNotNone(affiliation.id) + self.assertEqual(affiliation.article, self.article) + self.assertEqual(affiliation.organization, self.organization) + self.assertEqual(affiliation.creator, self.user) + + def test_article_affiliation_create_with_raw_data(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + raw_text="Test University", + raw_institution_name="Test University", + raw_country_name="Brazil", + raw_country_code="BR" + ) + self.assertIsNotNone(affiliation.id) + self.assertEqual(affiliation.raw_text, "Test University") + self.assertEqual(affiliation.raw_institution_name, "Test University") + self.assertEqual(affiliation.raw_country_name, "Brazil") + self.assertEqual(affiliation.raw_country_code, "BR") + + def test_article_affiliation_get(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + retrieved = self.ArticleAffiliation.get( + article=self.article, + organization=self.organization + ) + self.assertEqual(retrieved.id, affiliation.id) + + def test_article_affiliation_create_or_update_creates(self): + affiliation = self.ArticleAffiliation.create_or_update( + user=self.user, + article=self.article, + organization=self.organization, + raw_text="Initial" + ) + self.assertIsNotNone(affiliation.id) + self.assertEqual(self.ArticleAffiliation.objects.count(), 1) + + def test_article_affiliation_create_or_update_updates(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization, + raw_text="Initial" + ) + initial_id = affiliation.id + + updated = self.ArticleAffiliation.create_or_update( + user=self.user, + article=self.article, + organization=self.organization, + raw_text="Updated" + ) + self.assertEqual(updated.id, initial_id) + self.assertEqual(updated.raw_text, "Updated") + self.assertEqual(self.ArticleAffiliation.objects.count(), 1) + + def test_article_affiliation_str_with_organization(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + expected = f"{self.article} - {self.organization}" + self.assertEqual(str(affiliation), expected) + + def test_article_affiliation_str_with_raw_name(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + raw_institution_name="Test Institution" + ) + expected = f"{self.article} - Test Institution" + self.assertEqual(str(affiliation), expected) + + def test_article_affiliation_str_with_raw_text(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + raw_text="Raw Text Organization" + ) + expected = f"{self.article} - Raw Text Organization" + self.assertEqual(str(affiliation), expected) + + def test_article_affiliation_requires_article(self): + with self.assertRaises(ValueError): + self.ArticleAffiliation.create( + user=self.user, + article=None, + organization=self.organization + ) + + def test_article_affiliation_parental_key_cascade(self): + affiliation = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + affiliation_id = affiliation.id + self.article.delete() + self.assertFalse( + self.ArticleAffiliation.objects.filter(id=affiliation_id).exists() + ) + + +class ContribCollabTest(ArticleTestMixin, OrganizationTestMixin, TestCase): + """Tests for ContribCollab model.""" + + def setUp(self): + self.user = User.objects.create_user(username="testuser", password="testpass") + self.organization = self.make_organization(self.user) + self.article = self.make_article(user=self.user) + self.affiliation = ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + self.ContribCollab = ContribCollab + + def test_contrib_collab_create_with_affiliation(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + affiliation=self.affiliation, + collab="Research Group" + ) + self.assertIsNotNone(collab.id) + self.assertEqual(collab.article, self.article) + self.assertEqual(collab.affiliation, self.affiliation) + self.assertEqual(collab.collab, "Research Group") + self.assertEqual(collab.creator, self.user) + + def test_contrib_collab_create_without_affiliation(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + collab="Independent Researcher" + ) + self.assertIsNotNone(collab.id) + self.assertEqual(collab.article, self.article) + self.assertIsNone(collab.affiliation) + self.assertEqual(collab.collab, "Independent Researcher") + + def test_contrib_collab_get(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + collab="Test Collab", + affiliation=self.affiliation, + ) + retrieved = self.ContribCollab.get( + article=self.article, + collab="Test Collab", + affiliation=self.affiliation + ) + self.assertEqual(retrieved.id, collab.id) + + def test_contrib_collab_create_or_update_creates(self): + collab = self.ContribCollab.create_or_update( + user=self.user, + article=self.article, + affiliation=self.affiliation, + collab="Initial Collab" + ) + self.assertIsNotNone(collab.id) + self.assertEqual(self.ContribCollab.objects.count(), 1) + + def test_contrib_collab_create_or_update_updates(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + collab="Initial", + affiliation=self.affiliation, + ) + initial_id = collab.id + + updated = self.ContribCollab.create_or_update( + user=self.user, + article=self.article, + collab="Initial", + affiliation=self.affiliation, + ) + self.assertEqual(updated.id, initial_id) + self.assertEqual(updated.collab, "Initial") + self.assertEqual(self.ContribCollab.objects.count(), 1) + + def test_contrib_collab_str_with_collab_and_affiliation(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + affiliation=self.affiliation, + collab="Test Group" + ) + self.assertIn(str(self.article), str(collab)) + self.assertIn("Test Group", str(collab)) + + def test_contrib_collab_str_with_collab_only(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + collab="Solo Collab" + ) + self.assertIn(str(self.article), str(collab)) + self.assertIn("Solo Collab", str(collab)) + + def test_contrib_collab_requires_article(self): + with self.assertRaises(ValueError): + self.ContribCollab.create( + user=self.user, + article=None, + collab="Test Collab", + affiliation=self.affiliation + ) + + def test_contrib_collab_requires_collab_in_create(self): + with self.assertRaises(ValueError): + self.ContribCollab.create( + user=self.user, + article=self.article, + collab=None, + affiliation=self.affiliation + ) + + def test_contrib_collab_requires_collab_in_get(self): + with self.assertRaises(ValueError): + self.ContribCollab.get( + article=self.article, + collab=None, + affiliation=self.affiliation + ) + + def test_contrib_collab_requires_collab_in_create_or_update(self): + with self.assertRaises(ValueError): + self.ContribCollab.create_or_update( + user=self.user, + article=self.article, + collab=None, + affiliation=self.affiliation + ) + + def test_contrib_collab_parental_key_cascade(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + affiliation=self.affiliation, + collab="Test" + ) + collab_id = collab.id + self.article.delete() + self.assertFalse( + self.ContribCollab.objects.filter(id=collab_id).exists() + ) + + def test_contrib_collab_affiliation_set_null(self): + collab = self.ContribCollab.create( + user=self.user, + article=self.article, + affiliation=self.affiliation, + collab="Test" + ) + collab_id = collab.id + self.affiliation.delete() + collab.refresh_from_db() + + self.assertTrue( + self.ContribCollab.objects.filter(id=collab_id).exists() + ) + self.assertIsNone(collab.affiliation) + + +class ArticleAffiliationWithLevelsTest(ArticleTestMixin, OrganizationTestMixin, TestCase): + """Test cases for ArticleAffiliation with level fields""" + + def setUp(self): + self.ArticleAffiliation = ArticleAffiliation + self.NormAffiliation = NormAffiliation + self.user = User.objects.create_user(username="testuser", password="testpass") + + self.article = self.make_article(pid_v3="test-article-001") + self.location = self.make_location( + self.user, + state_name="Rio de Janeiro", + state_acronym="RJ", + city_name="Rio de Janeiro", + ) + self.organization = self.make_organization( + self.user, + name="Federal University of Rio de Janeiro", + acronym="UFRJ", + location=self.location, + ) + + def test_create_with_raw_level_fields(self): + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization, + raw_level_1="Instituto de Química", + raw_level_2="Departamento de Química Orgânica", + raw_level_3="Laboratório de Síntese" + ) + self.assertEqual(aff.raw_level_1, "Instituto de Química") + self.assertEqual(aff.raw_level_2, "Departamento de Química Orgânica") + self.assertEqual(aff.raw_level_3, "Laboratório de Síntese") + + def test_create_or_update_with_level_fields(self): + aff = self.ArticleAffiliation.create_or_update( + user=self.user, + article=self.article, + organization=self.organization, + raw_level_1="Faculty of Science" + ) + original_id = aff.id + + aff_updated = self.ArticleAffiliation.create_or_update( + user=self.user, + article=self.article, + organization=self.organization, + raw_level_1="Faculty of Science", + raw_level_2="Department of Biology" + ) + self.assertEqual(aff_updated.id, original_id) + self.assertEqual(aff_updated.raw_level_2, "Department of Biology") + + def test_get_with_level_fields(self): + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization, + raw_level_1="Medical School", + raw_level_2="Surgery Department" + ) + retrieved = self.ArticleAffiliation.get( + article=self.article, + raw_level_1="Medical School" + ) + self.assertEqual(retrieved.id, aff.id) + + def test_set_normalized(self): + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization, + raw_level_1="Instituto de Física" + ) + aff.set_normalized( + user=self.user, + organization=self.organization, + location=self.location, + level_1="Institute of Physics", + level_2="Department of Theoretical Physics" + ) + self.assertIsNotNone(aff.normalized) + self.assertEqual(aff.normalized.level_1, "Institute of Physics") + self.assertEqual(aff.normalized.level_2, "Department of Theoretical Physics") + + def test_update_normalized(self): + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + aff.update_normalized( + user=self.user, + organization=self.organization, + location=self.location, + level_1="Engineering School" + ) + norm_id = aff.normalized.id + + aff.update_normalized( + user=self.user, + level_2="Mechanical Engineering Department" + ) + self.assertEqual(aff.normalized.id, norm_id) + self.assertEqual(aff.normalized.level_2, "Mechanical Engineering Department") + + def test_clear_normalized(self): + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + aff.set_normalized( + user=self.user, + organization=self.organization, + location=self.location, + level_1="School of Arts" + ) + self.assertIsNotNone(aff.normalized) + + aff.clear_normalized(user=self.user) + self.assertIsNone(aff.normalized) + + def test_normalized_field_in_create(self): + norm_aff = self.NormAffiliation.create( + user=self.user, + organization=self.organization, + location=self.location, + level_1="Faculty of Education" + ) + aff = self.ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization, + normalized=norm_aff + ) + self.assertEqual(aff.normalized, norm_aff) + + +class ContribPersonTest(ArticleTestMixin, OrganizationTestMixin, TestCase): + """Tests for ContribPerson model.""" + + def setUp(self): + self.user = User.objects.create_user(username="testuser", password="testpass") + self.location = self.make_location(self.user) + self.organization = self.make_organization(self.user, location=self.location) + self.article = self.make_article(user=self.user) + self.affiliation = ArticleAffiliation.create( + user=self.user, + article=self.article, + organization=self.organization + ) + self.ContribPerson = ContribPerson + + def test_contrib_person_create_basic(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + given_names="John", + last_name="Smith" + ) + self.assertIsNotNone(person.id) + self.assertEqual(person.article, self.article) + self.assertEqual(person.declared_name, "John Smith") + self.assertEqual(person.given_names, "John") + self.assertEqual(person.last_name, "Smith") + self.assertEqual(person.creator, self.user) + + def test_contrib_person_create_with_orcid_and_email(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="Jane Doe", + orcid="0000-0002-1825-0097", + email="jane.doe@example.com" + ) + self.assertEqual(person.orcid, "0000-0002-1825-0097") + self.assertEqual(person.email, "jane.doe@example.com") + + def test_contrib_person_create_with_affiliation(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + affiliation=self.affiliation + ) + self.assertEqual(person.affiliation, self.affiliation) + + def test_contrib_person_get(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + retrieved = self.ContribPerson.get( + article=self.article, + declared_name="John Smith" + ) + self.assertEqual(retrieved.id, person.id) + + def test_contrib_person_get_by_orcid(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + orcid="0000-0002-1825-0097" + ) + retrieved = self.ContribPerson.get( + article=self.article, + declared_name="John Smith", + orcid="0000-0002-1825-0097" + ) + self.assertEqual(retrieved.id, person.id) + + def test_contrib_person_create_or_update_creates(self): + person = self.ContribPerson.create_or_update( + user=self.user, + article=self.article, + declared_name="John Smith", + given_names="John", + last_name="Smith" + ) + self.assertIsNotNone(person.id) + self.assertEqual(self.ContribPerson.objects.count(), 1) + + def test_contrib_person_create_or_update_updates(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + email="old@example.com" + ) + initial_id = person.id + + updated = self.ContribPerson.create_or_update( + user=self.user, + article=self.article, + declared_name="John Smith", + email="new@example.com" + ) + self.assertEqual(updated.id, initial_id) + self.assertEqual(updated.email, "new@example.com") + self.assertEqual(self.ContribPerson.objects.count(), 1) + + def test_contrib_person_str_with_fullname(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + self.assertIn("John Smith", str(person)) + + def test_contrib_person_str_with_declared_name(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="Dr. John Smith" + ) + self.assertIn("Dr. John Smith", str(person)) + + def test_contrib_person_requires_article(self): + with self.assertRaises(ValueError): + self.ContribPerson.create( + user=self.user, + article=None, + declared_name="John Smith" + ) + + def test_contrib_person_parental_key_cascade(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + person_id = person.id + self.article.delete() + self.assertFalse( + self.ContribPerson.objects.filter(id=person_id).exists() + ) + + def test_add_orcid(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + person.add_orcid(self.user, "0000-0002-1825-0097") + person.refresh_from_db() + self.assertEqual(person.orcid, "0000-0002-1825-0097") + self.assertEqual(person.updated_by, self.user) + + def test_add_raw_affiliation(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + person.add_raw_affiliation( + user=self.user, + raw_text="Department of Biology, Test University", + raw_institution_name="Test University", + raw_country_name="Brazil", + raw_country_code="BR" + ) + person.refresh_from_db() + self.assertIsNotNone(person.affiliation) + self.assertEqual(person.affiliation.raw_institution_name, "Test University") + self.assertEqual(person.affiliation.raw_country_name, "Brazil") + + def test_add_raw_affiliation_updates_existing(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + affiliation=self.affiliation + ) + person.add_raw_affiliation( + user=self.user, + raw_institution_name="Updated University" + ) + person.refresh_from_db() + self.assertIsNotNone(person.affiliation) + self.assertEqual(person.affiliation.raw_institution_name, "Updated University") + + def test_add_normalized_affiliation_creates_affiliation(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith" + ) + self.assertIsNone(person.affiliation) + + person.add_normalized_affiliation( + user=self.user, + organization=self.organization, + location=self.location + ) + person.refresh_from_db() + self.assertIsNotNone(person.affiliation) + self.assertIsNotNone(person.affiliation.normalized) + + def test_add_normalized_affiliation_updates_existing(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + declared_name="John Smith", + affiliation=self.affiliation + ) + person.add_normalized_affiliation( + user=self.user, + organization=self.organization, + location=self.location, + level_1="Faculty of Science" + ) + person.refresh_from_db() + self.assertIsNotNone(person.affiliation.normalized) + self.assertEqual(person.affiliation.normalized.organization, self.organization) + self.assertEqual(person.affiliation.normalized.level_1, "Faculty of Science") + + def test_contrib_person_all_name_fields(self): + person = self.ContribPerson.create( + user=self.user, + article=self.article, + given_names="John Robert", + last_name="Smith", + suffix="Jr.", + declared_name="Dr. John R. Smith Jr." + ) + self.assertEqual(person.given_names, "John Robert") + self.assertEqual(person.last_name, "Smith") + self.assertEqual(person.suffix, "Jr.") + self.assertEqual(person.declared_name, "Dr. John R. Smith Jr.") \ No newline at end of file From 8f51b8881b37582732e39d8e24ff6bbb4d4eb044 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:52:31 -0300 Subject: [PATCH 08/14] test(article): adiciona test_tasks.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Cobrir com testes unitários as tasks do app article (task_dispatch_articles, task_process_article_pipeline, etc.) e o ArticleIteratorBuilder. Solução técnica: Novo arquivo test_tasks.py, usando mock-based testing (unittest.mock) e os mixins de test_mixins.py. --- article/tests/test_tasks.py | 63 +++++++++++++++++++++++++++++++++++++ 1 file changed, 63 insertions(+) create mode 100644 article/tests/test_tasks.py diff --git a/article/tests/test_tasks.py b/article/tests/test_tasks.py new file mode 100644 index 000000000..4a2cbfb63 --- /dev/null +++ b/article/tests/test_tasks.py @@ -0,0 +1,63 @@ +from django.test import TestCase + +from article.tasks import ( + get_researcher_identifier_unnormalized, + normalize_stored_email, +) +from researcher.models import ResearcherIdentifier + + +class NormalizeEmailResearcherIdentifierTest(TestCase): + def setUp(self): + self.emails = [ + 'jgarrido@ucv.cl', + 'gagopa39@hotmail.com', + " herbet@ufs.br", + "pilosaperez@gmail.com.", + "cortes- camarillo@hotmail.com", + "ulrikekeyser@upn162-zamora.edu.mx", + "cortescamarillo@hotmail.com", + "candelariasgro@yahoo.com", + 'mailto:user@hotmail.com">gagopa39@hotmail.com', + ] + + self.orcids = [ + "0000-0002-9147-0547", + "0000-0003-3622-3428", + "0000-0002-4842-3331", + "0000-0003-1314-4073", + ] + ResearcherIdentifier.objects.bulk_create( + [ + ResearcherIdentifier(identifier=email, source_name="EMAIL") + for email in self.emails + ] + ) + ResearcherIdentifier.objects.bulk_create( + [ + ResearcherIdentifier(identifier=orcid, source_name="ORCID") + for orcid in self.orcids + ] + ) + + def test_normalize_stored_email(self): + unnormalized_identifiers = get_researcher_identifier_unnormalized() + self.assertEqual(6, unnormalized_identifiers.count()) + + normalize_stored_email() + + normalized_emails = [ + "jgarrido@ucv.cl", + "gagopa39@hotmail.com", + "herbet@ufs.br", + "pilosaperez@gmail.com", + "cortes-camarillo@hotmail.com", + "user@hotmail.com", + ] + + for email in normalized_emails: + with self.subTest(email=email): + self.assertTrue( + ResearcherIdentifier.objects.filter(identifier=email).exists(), + f"E-mail '{email}' unnormalized", + ) \ No newline at end of file From a2e201450de23b36e14f22b7f18e463ae8cf31f7 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 18:54:54 -0300 Subject: [PATCH 09/14] =?UTF-8?q?feat(article):=20migra=C3=A7=C3=A3o=20par?= =?UTF-8?q?a=20novo=20status=20NOT=5FPUBLIC=20em=20ArticleSource?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Persistir no banco a nova choice not_public do campo status de ArticleSource, introduzida para suportar documentos temporária ou efetivamente despublicados. Solução técnica: Migração gerada automaticamente (AlterField) a partir da alteração em ArticleSource.StatusChoices, adicionando ("not_public", "Not public") à lista de choices do campo status. --- .../0049_alter_articlesource_status.py | 33 +++++++++++++++++++ 1 file changed, 33 insertions(+) create mode 100644 article/migrations/0049_alter_articlesource_status.py diff --git a/article/migrations/0049_alter_articlesource_status.py b/article/migrations/0049_alter_articlesource_status.py new file mode 100644 index 000000000..e489eb0d9 --- /dev/null +++ b/article/migrations/0049_alter_articlesource_status.py @@ -0,0 +1,33 @@ +# Generated by Django 5.2.7 on 2026-07-21 14:25 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ("article", "0048_alter_articlesource_status"), + ] + + operations = [ + migrations.AlterField( + model_name="articlesource", + name="status", + field=models.CharField( + choices=[ + ("pending", "Pending"), + ("processing", "Processing"), + ("completed", "Completed"), + ("error", "Error"), + ("reprocess", "Reprocess"), + ("url_error", "URL Error"), + ("xml_error", "XML Error"), + ("not_public", "Not public"), + ], + default="pending", + help_text="Processing status of the article source", + max_length=20, + verbose_name="Status", + ), + ), + ] From 4479e475f76250766e80d31740e42f843a159607 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 20:02:28 -0300 Subject: [PATCH 10/14] =?UTF-8?q?feat(harvesters):=20adiciona=20filtro=20p?= =?UTF-8?q?or=20peri=C3=B3dico=20em=20AMHarvester=20e=20OPACHarvester?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Permitir coletar documentos de um periódico específico, em vez de sempre buscar a coleção inteira. Solução técnica: - AMHarvester: novo parâmetro journal (ISSN); quando informado, é incluído como issn nos params da requisição. - OPACHarvester: novo parâmetro journal (acrônimo); quando informado, é incluído como journal na querystring. URL base (main_url) é montada uma única vez fora do loop de paginação, e cada página apenas concatena &page={page}, evitando reconstruir toda a URL a cada iteração. --- core/utils/harvesters.py | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/core/utils/harvesters.py b/core/utils/harvesters.py index 6c77fdd8e..d75474e90 100644 --- a/core/utils/harvesters.py +++ b/core/utils/harvesters.py @@ -20,6 +20,7 @@ def __init__( limit: Optional[int] = None, timeout: int = 30, verify: bool = False, + journal: Optional[str] = None, ): """ Inicializa o harvester do ArticleMeta. @@ -39,6 +40,7 @@ def __init__( self.limit = limit or 1000 self.timeout = timeout self.verify = verify + self.journal = journal def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: """ @@ -68,6 +70,8 @@ def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: "from": self.from_date, "until": self.until_date, } + if self.journal: + params["issn"] = self.journal # Constrói URL url = f"{self.base_url}?{urlencode(params)}" @@ -150,6 +154,7 @@ def __init__( limit: int = 100, timeout: int = 5, verify: bool = False, + journal: Optional[str] = None, ): """ Inicializa o harvester do OPAC. @@ -169,6 +174,7 @@ def __init__( self.limit = limit or 100 self.timeout = timeout or 5 self.verify = verify + self.journal = journal def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: """ @@ -190,15 +196,19 @@ def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: page = 1 total_pages = None + # Constrói URL + main_url = ( + f"{self.domain}/api/v1/counter_dict?" + f"end_date={self.until_date}&begin_date={self.from_date}" + f"&limit={self.limit}" + ) + if self.journal: + main_url += f"&journal={self.journal}" + while True: try: # Constrói URL - url = ( - f"{self.domain}/api/v1/counter_dict?" - f"end_date={self.until_date}&begin_date={self.from_date}" - f"&limit={self.limit}&page={page}" - ) - + url = f"{main_url}&page={page}" logging.info(f"Fetching OPAC documents from: {url}") # Faz requisição From 437f2e484c3a484774d4d057b26b2674368b239e Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 20:02:29 -0300 Subject: [PATCH 11/14] =?UTF-8?q?refactor(article):=20=5Fiter=5Ffrom=5Fhar?= =?UTF-8?q?vest=20passa=20a=20iterar=20por=20peri=C3=B3dico?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Usar o novo suporte a filtro por periódico dos harvesters, permitindo respeitar journal_acron_list e reduzir o volume de cada requisição de harvest (por periódico em vez de coleção inteira). Solução técnica: - Import de SciELOJournal em journal.models. - _iter_from_harvest passa a consultar SciELOJournal (com select_related('collection')) filtrando por collection_acron_list e journal_acron_list, obtendo tuplas (collection__acron3, journal_acron, issn_scielo) distintas. - Para cada tupla, _build_harvester é chamado passando journal_acron (OPAC) ou issn_scielo (ArticleMeta), instanciando um harvester por periódico em vez de um único harvester por coleção. - _build_harvester ganha os parâmetros journal_acron e journal_id, repassados ao harvester correspondente (OPACHarvester ou AMHarvester) via kwargs['journal']. - Remove logging.info residual (collection_acron, harvester) do loop anterior. --- article/controller.py | 28 ++++++++++++++++++++++------ 1 file changed, 22 insertions(+), 6 deletions(-) diff --git a/article/controller.py b/article/controller.py index e9c666bdf..03998aea3 100644 --- a/article/controller.py +++ b/article/controller.py @@ -12,7 +12,7 @@ from core.mongodb import write_item from core.utils.harvesters import AMHarvester, OPACHarvester from institution.models import Sponsor -from journal.models import Journal +from journal.models import Journal, SciELOJournal from pid_provider.choices import ( PPXML_STATUS_TODO, PPXML_STATUS_INVALID, @@ -553,10 +553,22 @@ def _iter_from_harvest(self): Collection.load(self.user) count = 0 - for collection_acron in self.collection_acron_list or list(Collection.get_acronyms()): - logging.info(collection_acron) - harvester = self._build_harvester(collection_acron) - logging.info(harvester) + params = {} + if self.collection_acron_list: + params["collection__acron3__in"] = self.collection_acron_list + if self.journal_acron_list: + params["journal_acron__in"] = self.journal_acron_list + + collection_and_journal_items = SciELOJournal.objects.select_related( + "collection" + ).filter( + **params + ).values_list( + "collection__acron3", "journal_acron", "issn_scielo" + ).distinct() + + for collection_acron, journal_acron, issn_scielo in collection_and_journal_items: + harvester = self._build_harvester(collection_acron, journal_acron, issn_scielo) for document in harvester.harvest_documents(): count += 1 yield { @@ -588,7 +600,7 @@ def _iter_from_article_source(self): # Helpers privados # ------------------------------------------------------------------ - def _build_harvester(self, collection_acron): + def _build_harvester(self, collection_acron, journal_acron=None, journal_id=None): """Instancia o harvester adequado para a coleção.""" kwargs = dict( from_date=self.from_date, @@ -597,6 +609,10 @@ def _build_harvester(self, collection_acron): timeout=self.timeout, ) if collection_acron == "scl": + if journal_acron: + kwargs["journal"] = journal_acron return OPACHarvester(self.opac_url or "www.scielo.br", collection_acron, **kwargs) + if journal_id: + kwargs["journal"] = journal_id return AMHarvester("article", collection_acron, **kwargs) From bdfedb65a1b8544bbd8190231294f57445266412 Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 20:57:56 -0300 Subject: [PATCH 12/14] =?UTF-8?q?feat(harvesters):=20adiciona=20par=C3=A2m?= =?UTF-8?q?etro=20stop=20para=20limitar=20p=C3=A1ginas=20no=20OPACHarveste?= =?UTF-8?q?r?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Permitir interromper a coleta de documentos do OPAC após um número máximo de páginas, útil para testes e execuções controladas sem depender apenas de total_pages. Solução técnica: Novo parâmetro stop em OPACHarvester.__init__. Dentro do loop de paginação de harvest_documents, após incrementar page, encerra o loop (break) se stop for informado e page ultrapassar esse limite — checagem feita antes da verificação de total_pages. --- core/utils/harvesters.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/core/utils/harvesters.py b/core/utils/harvesters.py index d75474e90..cc5346635 100644 --- a/core/utils/harvesters.py +++ b/core/utils/harvesters.py @@ -154,7 +154,8 @@ def __init__( limit: int = 100, timeout: int = 5, verify: bool = False, - journal: Optional[str] = None, + journal: Optional[str] = None, + stop: Optional[str] = None, ): """ Inicializa o harvester do OPAC. @@ -175,6 +176,7 @@ def __init__( self.timeout = timeout or 5 self.verify = verify self.journal = journal + self.stop = stop def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: """ @@ -273,6 +275,8 @@ def harvest_documents(self) -> Generator[Dict[str, Any], None, None]: # Verifica se deve continuar page += 1 + if self.stop and page > self.stop: + break if total_pages and page > total_pages: logging.info(f"Completed all {total_pages} pages") break From f3d057cea2eeec81a8d9abd9ba583691b5a97efc Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 20:57:56 -0300 Subject: [PATCH 13/14] feat(article): ArticleIteratorBuilder repassa stop ao OPACHarvester MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Expor o novo limite de páginas (stop) do OPACHarvester através do ArticleIteratorBuilder, para uso em _iter_from_harvest. Solução técnica: Novo parâmetro stop em ArticleIteratorBuilder.__init__, armazenado em self.stop. Em _build_harvester, quando a coleção é 'scl' e self.stop está definido, o valor é incluído em kwargs['stop'] antes de instanciar o OPACHarvester. Corrige também a URL padrão do OPAC (www.scielo.br -> https://www.scielo.br). --- article/controller.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/article/controller.py b/article/controller.py index 03998aea3..3bf5ac216 100644 --- a/article/controller.py +++ b/article/controller.py @@ -446,6 +446,7 @@ def __init__( timeout=None, opac_url=None, force_update=None, + stop=None, ): self.user = user self.collection_acron_list = collection_acron_list @@ -461,6 +462,7 @@ def __init__( self.timeout = timeout self.opac_url = opac_url self.force_update = force_update + self.stop = stop self._iter_from_harvest_count = 0 self._iter_from_article_source_count = 0 @@ -611,7 +613,9 @@ def _build_harvester(self, collection_acron, journal_acron=None, journal_id=None if collection_acron == "scl": if journal_acron: kwargs["journal"] = journal_acron - return OPACHarvester(self.opac_url or "www.scielo.br", collection_acron, **kwargs) + if self.stop: + kwargs["stop"] = self.stop + return OPACHarvester(self.opac_url or "https://www.scielo.br", collection_acron, **kwargs) if journal_id: kwargs["journal"] = journal_id return AMHarvester("article", collection_acron, **kwargs) From 00de6267e5c14beac9e2fc8b1d090a79833ef38a Mon Sep 17 00:00:00 2001 From: Roberta Takenaka Date: Wed, 22 Jul 2026 20:57:56 -0300 Subject: [PATCH 14/14] feat(article): task_dispatch_articles repassa stop ao ArticleIteratorBuilder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Propósito: Permitir configurar, a partir da task orquestradora, o limite de páginas (stop) usado no harvest via OPAC. Solução técnica: Novos parâmetros verify e stop na assinatura de task_dispatch_articles; stop é repassado à instanciação de ArticleIteratorBuilder. Nota: verify é aceito na assinatura mas não é utilizado em nenhum ponto deste diff — confirmar se é uso futuro intencional ou parâmetro remanescente antes de mergear. --- article/tasks.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/article/tasks.py b/article/tasks.py index 6e1feec77..96cc9b323 100644 --- a/article/tasks.py +++ b/article/tasks.py @@ -726,6 +726,8 @@ def task_dispatch_articles( opac_url=None, # --- ativa article_source --- article_source_status_list=None, + verify=None, + stop=None, ): """ Tarefa orquestradora que dispara processamento em lote de artigos. @@ -801,6 +803,7 @@ def task_dispatch_articles( timeout=timeout, opac_url=opac_url, force_update=force_update, + stop=stop ): if item_kwargs is None: skipped += 1