diff --git a/cds_migration_progress.html b/cds_migration_progress.html new file mode 100644 index 00000000..e9046ed8 --- /dev/null +++ b/cds_migration_progress.html @@ -0,0 +1,862 @@ +CDS Migration Progress, 2015–2026 + + +
+
+
+

CDS migration progress, 2015–2026

+

Legacy CDS (cds.cern.ch) total record count (estimated including restricted records), plotted on the same scale as cumulative records migrated into repository.cern, cumulative books held in catalogue.library.cern since its 21 April 2021 launch, and cumulative videos held on videos.cern.ch since the 2017 migration.

+
+ +
+
+ Legacy CDS total records (est. incl. restricted) + repository.cern cumulative records + catalogue.library.cern cumulative records + videos.cern.ch cumulative records + Migration milestone +
+ +
+ +
+ +
+ View underlying data (Wayback Machine snapshots & migration events) + + + + +
Legacy CDS (cds.cern.ch) — Wayback Machine snapshots
Snapshot datePublic count (scraped)Retroactive adjustmentEst. total incl. restricted (plotted)repository.cern migrated to dateNet of repository.cern
+ + + + +
repository.cern — migration events
DateContent migratedRecords addedCumulative total
+ + + + +
catalogue.library.cern — books created per year (via /api/documents)
PeriodBooks createdCumulative total
+ + + + +
videos.cern.ch — videos created per year (via /api/records)
PeriodVideos createdCumulative total
+
+ + +
+
+ + diff --git a/cds_migrator_kit/rdm/migration_config.py b/cds_migrator_kit/rdm/migration_config.py index 727045bf..29ca7c73 100644 --- a/cds_migrator_kit/rdm/migration_config.py +++ b/cds_migrator_kit/rdm/migration_config.py @@ -535,7 +535,10 @@ def resolve_record_pid(pid): ### EP Approval configuration only needed for local, it should use cds-rdm config for de/sandbox/prod # =========================== -CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "78b3c4aa-c4e6-4502-8226-67ba2d347afe" +# ATTENTION: please don't modify this local value - the community is created +# via cds-rdm fixtures with this id - if you have another ID locally +# change it in your local db +CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" """The id of the CERN Scientific community. This is only a local-dev default: on other instances (sandbox/prod), set the @@ -626,4 +629,14 @@ def resolve_record_pid(pid): "counter_digits": 3, # zero-padding width, e.g. 3 → "001" }, }, + "6a289642-5378-4daf-87b5-bb58af00487a": { + # DIRAC + "label": "EP approval", # shown in UI buttons/headings + "referee_group": "cds-ph-ep-publications-referee-non-lhc", # CERN e-group slug + "report_number": { + "prefix": "CERN-EP", # literal prefix, e.g. "CERN-EP" + "include_year": True, # append the current year after prefix + "counter_digits": 3, # zero-padding width, e.g. 3 → "001" + }, + }, } diff --git a/cds_migrator_kit/rdm/records/load/load.py b/cds_migrator_kit/rdm/records/load/load.py index 4e18d000..86778c88 100644 --- a/cds_migrator_kit/rdm/records/load/load.py +++ b/cds_migrator_kit/rdm/records/load/load.py @@ -22,7 +22,7 @@ from invenio_db import db from invenio_db.uow import ModelCommitOp, UnitOfWork from invenio_i18n import _ -from invenio_pidstore.errors import PIDDoesNotExistError +from invenio_pidstore.errors import PIDAlreadyExists, PIDDoesNotExistError from invenio_pidstore.models import PersistentIdentifier from invenio_rdm_migrator.load.base import Load from invenio_rdm_records.proxies import current_rdm_records_service @@ -308,6 +308,22 @@ def _load(self, entry: MigrationEntry): # apply after record fully finished (does not sync at the spot, only enabled) self._apply_clc_sync(recid_state_after_load, entry) return recid_state_after_load + except PIDAlreadyExists: + # The legacy recid's `lrecid` PID was minted by someone else + # between our _should_skip_recid() check above and now - e.g. a + # concurrently running migration of another collection whose + # dump cross-lists the same legacy recid (two former-experiment + # collections can both ship the same record). Treat it exactly + # like _should_skip_recid: already migrated, nothing to do. + # this happens when you run several mirations at the same time + self.migration_logger.add_information( + recid, + state={ + "message": "Record already migrated (lrecid PID already minted)", + "value": recid, + }, + ) + self.migration_logger.finalise_record(recid) except (UnexpectedValue, ManualImportRequired, GrantCreationError) as e: self.migration_logger.add_log(e, record=entry) except (CDSMigrationException, ValidationError, InvalidRelationValue) as e: diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 2bca4a47..baa8985b 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -161,5 +161,15 @@ # Legacy experiment names remapped to vocabulary ids before lookup EXPERIMENT_ALIASES = { - "t2k": "re13", + "t2k": "RE13", + "antares": "RE6", + "dirac": "PS212", + "dirac ps212": "PS212", + "harp ps214": "PS214", + "harp": "PS214", + "dampe": "RE29", } + +# 693__e values that are keywords rather than real experiments (curated as +# "mots clef"), mapped to subjects instead of cern:experiments +EXPERIMENTS_AS_SUBJECTS = ["d3", "r104", "r105a"] diff --git a/cds_migrator_kit/rdm/records/transform/entities/record.py b/cds_migrator_kit/rdm/records/transform/entities/record.py index 4a574c91..9e386dd9 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/record.py +++ b/cds_migrator_kit/rdm/records/transform/entities/record.py @@ -168,7 +168,7 @@ def _files(self, dump): files = dump.files return {"enabled": bool(files)} - def _metadata(self, dojson_entry, raw_dump_entry): + def _metadata(self, dojson_entry, raw_dump_entry, pids=None): """Build the metadata dict by running the composed field mappers. Whether every ``dojson_entry`` key ended up consumed *somewhere* in @@ -183,6 +183,7 @@ def _metadata(self, dojson_entry, raw_dump_entry): raw_dump_entry=raw_dump_entry, migration_logger=self.migration_logger, affiliations_mapping=self.affiliations_mapping, + pids=pids or {}, ) metadata = ctx.metadata # Order matters: ResourceTypeMapper must run before TitleMapper reads @@ -259,10 +260,11 @@ def build(self) -> RecordBody: # same reason as _pids()/_access()'s record_restriction pop. internal_notes = dojson_entry.pop("internal_notes", None) + pids = self._pids(dojson_entry) record_json_output = { "files": self._files(dump), - "pids": self._pids(dojson_entry), - "metadata": self._metadata(dojson_entry, raw_dump_entry), + "pids": pids, + "metadata": self._metadata(dojson_entry, raw_dump_entry, pids), "internal_notes": internal_notes, "custom_fields": custom_fields, } diff --git a/cds_migrator_kit/rdm/records/transform/mappers/base.py b/cds_migrator_kit/rdm/records/transform/mappers/base.py index ec811f5f..506679bb 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/base.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/base.py @@ -28,6 +28,7 @@ class RecordTransformContext: migration_logger: object = None affiliations_mapping: object = None access_grants_view: object = None + pids: dict = field(default_factory=dict) metadata: dict = field(default_factory=dict) custom_fields: dict = field(default_factory=dict) diff --git a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py index 7150026b..84a1e56e 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py @@ -12,7 +12,10 @@ RecordFlaggedCuration, UnexpectedValue, ) -from cds_migrator_kit.rdm.records.transform.config import EXPERIMENT_ALIASES +from cds_migrator_kit.rdm.records.transform.config import ( + EXPERIMENT_ALIASES, + EXPERIMENTS_AS_SUBJECTS, +) from cds_migrator_kit.rdm.records.transform.mappers.base import CustomFieldMapper from cds_migrator_kit.rdm.records.transform.mappers.vocabulary import search_vocabulary @@ -34,6 +37,12 @@ def apply(self, ctx): for experiment in experiments: if experiment.lower().strip() in ["not applicable", "xx"]: continue + if experiment.lower().strip() in EXPERIMENTS_AS_SUBJECTS: + # curated as keywords ("mots clef"), not real experiments + ctx.dojson_entry.setdefault("subjects", []).append( + {"subject": experiment} + ) + continue experiment = EXPERIMENT_ALIASES.get(experiment.lower().strip(), experiment) result = search_vocabulary(experiment, "experiments") if result and result not in experiments_out: @@ -153,7 +162,12 @@ def apply(self, ctx): "cern:accelerators", [] ) for accelerator in accelerators: - if accelerator.lower().strip() in ["not applicable", "xx", "fermi"]: + if accelerator.lower().strip() in [ + "not applicable", + "xx", + "fermi", + "cern recognized expt.", + ]: continue result = search_vocabulary(accelerator, "accelerators") if result and result not in accelerators_out: diff --git a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py index f23f9f5d..ca00b2b1 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py @@ -189,12 +189,23 @@ def map_value(self, ctx): class TableOfContentsMapper(FieldMapper): - """Folds table_of_content into additional_descriptions.""" + """Folds table_of_content into additional_descriptions. + + Also the single place where the final ``additional_descriptions`` list + is deduplicated: many different dojson rules append to it (520/246/ + 035/500/210/... across base.py and the various collection-specific + rule modules), some legacy records repeat the very same MARC field + (identical text, sometimes only differing in a provenance subfield + nothing here reads), and not every one of those rules remembers to + guard against re-adding an entry already present. Deduplicating once + here, after every rule has run, doesn't depend on each of them getting + that guard right. + """ id = "additional_descriptions" def map_value(self, ctx): - """Move table_of_content into additional_descriptions and return it.""" + """Move table_of_content into additional_descriptions and dedupe.""" dojson_entry = ctx.dojson_entry toc = dojson_entry.get("table_of_content", []) additional_desc = dojson_entry.get("additional_descriptions", []) @@ -204,6 +215,14 @@ def map_value(self, ctx): ) dojson_entry["additional_descriptions"] = additional_desc dojson_entry.pop("table_of_content") + + deduped = [] + for description in dojson_entry.get("additional_descriptions", []): + if description not in deduped: + deduped.append(description) + if deduped: + dojson_entry["additional_descriptions"] = deduped + return dojson_entry.get("additional_descriptions") @@ -244,6 +263,37 @@ def map_value(self, ctx): return identifiers +#: `setlink` is a CDS-internal redirector, not a real related resource - +#: drop any related_identifiers entry pointing at it. +_SETLINK_URL_PREFIX = "http://documents.cern.ch/cgi-bin/setlink?" + + +class RelatedIdentifiersMapper(FieldMapper): + """Maps related_identifiers, dropping CDS-internal setlink URLs.""" + + id = "related_identifiers" + + def map_value(self, ctx): + """Return related_identifiers without setlink URLs or the record's own DOI.""" + related_identifiers = ctx.dojson_entry.get("related_identifiers", []) + record_doi = ((ctx.pids or {}).get("doi") or {}).get("identifier") + record_doi = record_doi.strip().lower() if record_doi else None + return [ + item + for item in related_identifiers + if not ( + (item.get("scheme") or "").upper() == "URL" + and (item.get("identifier") or "").startswith(_SETLINK_URL_PREFIX) + ) + # the record's own DOI is already in pids, don't repeat it + and not ( + record_doi + and (item.get("scheme") or "").lower() == "doi" + and (item.get("identifier") or "").strip().lower() == record_doi + ) + ] + + # Fields that pass through unchanged from dojson_entry - kept explicit in the # composed list (mappers/config equivalent) rather than open-ended, so the # "forgotten metadata key" completeness check in @@ -256,7 +306,6 @@ def map_value(self, ctx): "languages", "dates", "funding", - "related_identifiers", "rights", "copyright", ) diff --git a/cds_migrator_kit/rdm/records/transform/mappers/registry.py b/cds_migrator_kit/rdm/records/transform/mappers/registry.py index abe28f11..938d1cb7 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/registry.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/registry.py @@ -27,6 +27,7 @@ PASSTHROUGH_METADATA_FIELDS, IdentifiersMapper, PublicationDateMapper, + RelatedIdentifiersMapper, ResourceTypeMapper, SubjectsMapper, TableOfContentsMapper, @@ -45,6 +46,7 @@ PublicationDateMapper(), SubjectsMapper(), IdentifiersMapper(), + RelatedIdentifiersMapper(), *(PassthroughMapper(field_name) for field_name in PASSTHROUGH_METADATA_FIELDS), ) diff --git a/cds_migrator_kit/rdm/records/transform/models/lep.py b/cds_migrator_kit/rdm/records/transform/models/lep.py index 53fcadd4..46417bfd 100644 --- a/cds_migrator_kit/rdm/records/transform/models/lep.py +++ b/cds_migrator_kit/rdm/records/transform/models/lep.py @@ -20,6 +20,7 @@ class LEPResearchModel(ResearchModel): __ignore_keys__ = { "594__a", # can be ignored for this collection + "852__a", # location "775__p", # can be ignored for this collection - title of another volume "775__c", # year of volume "596__a", # multivolume tag diff --git a/cds_migrator_kit/rdm/records/transform/models/research.py b/cds_migrator_kit/rdm/records/transform/models/research.py index 4b8e0b78..59e2dd5b 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research.py +++ b/cds_migrator_kit/rdm/records/transform/models/research.py @@ -16,7 +16,7 @@ class ResearchModel(CdsOverdo): """Translation model for research.""" - __query__ = '693__.e:"DAMPE RE29" OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM' + __query__ = '693__.e:"DAMPE RE29" OR 693__.e:RE29 OR 693__.e:DAMPE OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM -980:BULLETINNEWS' __ignore_keys__ = { "0248_a", @@ -50,6 +50,7 @@ class ResearchModel(CdsOverdo): "542__8", # agreed not to migrate open access related fields "595__i", # TODO ?? "695__e", # some inspire tag + "695__9", # bibclassify "700__m", # email of contributor "700__q", # TODO ignore? aliteration of the name, used for searching "700__v", # TODO drop? diff --git a/cds_migrator_kit/rdm/records/transform/models/research_committee.py b/cds_migrator_kit/rdm/records/transform/models/research_committee.py index 3135e64b..13d08b02 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research_committee.py +++ b/cds_migrator_kit/rdm/records/transform/models/research_committee.py @@ -42,12 +42,15 @@ class ResearchCommitteeModel(CdsOverdo): "340__a", # TODO ignore material? "540__3", # TODO still ignore the material of the license? "542__3", # TODO still ignore the material of the license? + "594__a", # ATN tag "595__i", # TODO ?? "695__e", # some inspire tag + "695__9", # some inspire tag "700__m", # email of contributor "700__q", # TODO ignore? aliteration of the name, used for searching "700__v", # TODO drop? "773__x", # INSPIRE publication note + "852__a", "8564_8", # file id "8564_s", # bibdoc id "8564_x", # icon thumbnails sizes diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py index 695a4e3a..017a04cb 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py @@ -333,6 +333,9 @@ def report_number(self, key, value): raise IgnoreKey("related_identifiers") elif scheme.upper().startswith("B00"): raise IgnoreKey("related_identifiers") + elif key == "088__" and scheme.upper().startswith("SC000"): + # internal scanning/digitisation request number, to drop + raise IgnoreKey("related_identifiers") elif scheme.startswith("SCOO"): identifier = scheme scheme = "other" @@ -828,6 +831,20 @@ def series_information(self, key, value): return {"description": series, "type": {"id": "series-information"}} +@model.over("additional_descriptions", "^336__") +@for_each_value +def multiple_videos_note(self, key, value): + """Translate the video-system's "multiple videos" cross-reference note. + + The only recognised 336__a content - any other value is unexpected and + flagged for manual curation rather than silently dropped. + """ + note = StringValue(value.get("a", "")).parse() + if not note.startswith("Multiple videos have been identified with recid"): + raise UnexpectedValue(field=key, subfield="a", value=value) + return {"description": note, "type": {"id": "technical-info"}} + + @model.over("related_identifiers", "^084__") @for_each_value def yellow_reports(self, key, value): @@ -924,13 +941,17 @@ def related_identifiers_787(self, key, value): "resource_type": {"id": "publication-report"}, }, "complemented by": { - "relation_type": {"id": "issuplementedby"}, + "relation_type": {"id": "issupplementedby"}, "resource_type": {"id": "publication-report"}, }, "preprint": { "relation_type": {"id": "references"}, "resource_type": {"id": "publication-preprint"}, }, + "related video": { + "relation_type": {"id": "references"}, + "resource_type": {"id": "video"}, + }, } if recid: diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py index e1ac8b33..c9c444aa 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py @@ -748,6 +748,9 @@ def resource_type(self, key, value): "lhcf_proc": {"id": "publication-conferenceproceeding"}, "lhcf_reports": {"id": "publication-report"}, "conferencepapers": {"id": "publication-conferencepaper"}, + "technical note": {"id": "publication-technicalnote"}, + "minutes": {"id": "publication-meetingminutes"}, + "presentation": {"id": "presentation"}, } try: diff --git a/cds_migrator_kit/rdm/streams.yaml b/cds_migrator_kit/rdm/streams.yaml index 5199d992..c8a9dc2e 100644 --- a/cds_migrator_kit/rdm/streams.yaml +++ b/cds_migrator_kit/rdm/streams.yaml @@ -141,9 +141,9 @@ records: - "28bf99b9-1e72-405a-b95a-828019831def" - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" scp_adv: - data_dir: cds_migrator_kit/rdm/data/committees/sc_sdv + data_dir: cds_migrator_kit/rdm/data/committees/scp_sdv extract: - dirpath: cds_migrator_kit/rdm/data/committees/sc_adv + dirpath: cds_migrator_kit/rdm/data/committees/scp_adv transform: files_dump_dir: cds_migrator_kit/rdm/data/committees/files/ missing_users: cds_migrator_kit/rdm/data/users @@ -237,33 +237,10 @@ records: communities_ids: - "88a5cdf4-974a-44d4-b145-d19ee6346bb8" - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" - lcd_restr: - plots: true - data_dir: cds_migrator_kit/rdm/data/former_exp/lcd_restr - restricted: "True" - create_inclusion_request: true - extract: - dirpath: cds_migrator_kit/rdm/data/former_exp/lcd_restr - transform: - files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ - missing_users: cds_migrator_kit/rdm/data/users - communities_ids: - - "adc02716-780b-4bba-8e89-6316c9a11cf0" - lcd: - plots: true - create_inclusion_request: true - data_dir: cds_migrator_kit/rdm/data/former_exp/lcd - extract: - dirpath: cds_migrator_kit/rdm/data/former_exp/lcd - transform: - files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ - missing_users: cds_migrator_kit/rdm/data/users - communities_ids: - - "adc02716-780b-4bba-8e89-6316c9a11cf0" - - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" re29: data_dir: cds_migrator_kit/rdm/data/former_exp/re29 plots: true + create_inclusion_request: true extract: dirpath: cds_migrator_kit/rdm/data/former_exp/re29 transform: @@ -290,6 +267,7 @@ records: data_dir: cds_migrator_kit/rdm/data/former_exp/ua2 extract: dirpath: cds_migrator_kit/rdm/data/former_exp/ua2 + preferred_model: research transform: files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ missing_users: cds_migrator_kit/rdm/data/users diff --git a/cds_migrator_kit/rdm/streams_shelved.yaml b/cds_migrator_kit/rdm/streams_shelved.yaml index bd155cc1..371e50a6 100644 --- a/cds_migrator_kit/rdm/streams_shelved.yaml +++ b/cds_migrator_kit/rdm/streams_shelved.yaml @@ -1,16 +1,29 @@ db_uri: postgresql://cds-rdm:cds-rdm@localhost:5432/cds-rdm records: - thesis: - data_dir: cds_migrator_kit/rdm/data/thesis + lcd_restr: + plots: true + data_dir: cds_migrator_kit/rdm/data/former_exp/lcd_restr + restricted: "True" + create_inclusion_request: true extract: - dirpath: cds_migrator_kit/rdm/data/thesis/dump/ + dirpath: cds_migrator_kit/rdm/data/former_exp/lcd_restr transform: - files_dump_dir: cds_migrator_kit/rdm/data/thesis/files/ + files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f - load: - legacy_pids_to_redirect: cds_migrator_kit/rdm/data/thesis/duplicated_pids.json + - "adc02716-780b-4bba-8e89-6316c9a11cf0" + lcd: + plots: true + create_inclusion_request: true + data_dir: cds_migrator_kit/rdm/data/former_exp/lcd + extract: + dirpath: cds_migrator_kit/rdm/data/former_exp/lcd + transform: + files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ + missing_users: cds_migrator_kit/rdm/data/users + communities_ids: + - "adc02716-780b-4bba-8e89-6316c9a11cf0" + - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" mous: data_dir: cds_migrator_kit/rdm/data/mous extract: diff --git a/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py b/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py index a10886e2..c02cb438 100644 --- a/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py +++ b/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py @@ -146,7 +146,7 @@ class SubmitterModel(CdsOverdo): "542__u", # https://cds.cern.ch/record/2285212/export/hm?ln=en "560172", # https://cds.cern.ch/record/383486/export/hm?ln=en "56017a", # https://cds.cern.ch/record/383486/export/hm?ln=en wrong keyword subfield - "506__m", # mail + # "506__m", # e-group/reader's email, used to find/recreate their account "590__b", # abstract translation "590__a", # abstract translation TODO https://cds.cern.ch/record/1476067/export/hm?ln=en "594__a", # https://cds.cern.ch/record/466504/export/hm?ln=en, 455788 @@ -366,3 +366,9 @@ class SubmitterModel(CdsOverdo): bases=(base_model,), entry_point_group="cds_migrator_kit.migrator.rules.submitter", ) + +# Registers the 506 access-grant-emails rule directly on submitter_model, +# after it has been built above - see access_grants.py's module docstring +# for why this must be isolated to this instance instead of the shared +# base_model. +import cds_migrator_kit.rdm.users.transform.xml_processing.rules.access_grants # noqa: E402,F401 diff --git a/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py b/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py new file mode 100644 index 00000000..db87e021 --- /dev/null +++ b/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py @@ -0,0 +1,54 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""CDS-RDM access grant accounts migration rules. + +Mirrors the 859__f "submitter" / 906__m "reviewer" rules (see +cds_migrator_kit/transform/xml_processing/rules/base.py and +cds_migrator_kit/rdm/users/transform/xml_processing/rules/reviewers.py): +506 access restriction fields (see the `access_grants` rule in +cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py) +can name a person directly by email, in subfields d/m/a, instead of an +e-group/role name. Those emails need an account pre-created too, so the +record's actual access grant can later be resolved to a `User` +(cds_migrator_kit/rdm/records/transform/entities/parent.py, `resolve_grants`). + +Registered directly on `submitter_model`, not on the shared `base_model`: +research.py/hr.py/it.py/faser_publication.py already register their own, +unrelated "^506[1_]_" rule on their own separate model instances, so this +rule must stay isolated to `submitter_model` to avoid clashing with those. +This is why it is imported at the bottom of +cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py, +after `submitter_model` has been constructed. +""" + +import re + +from dojson.errors import IgnoreKey + +from cds_migrator_kit.rdm.users.transform.xml_processing.models.submitter import ( + submitter_model, +) + +EMAIL_PATTERN = re.compile(r"[^@]+@[^@]+\.[^@]+") + + +@submitter_model.over("access_grant_emails", "^506[1_]_") +def record_access_grant_emails(self, key, value): + """Translate 506 access grant emails, ignoring e-group/role names.""" + emails = self.get("access_grant_emails", []) + for subfield in ("d", "m", "a"): + raw = value.get(subfield) + if isinstance(raw, tuple): + raw = raw[0] + if not raw: + continue + candidate = raw.strip().lower() + if EMAIL_PATTERN.match(candidate) and candidate not in emails: + emails.append(candidate) + self["access_grant_emails"] = emails + raise IgnoreKey("access_grant_emails") diff --git a/cds_migrator_kit/users/load.py b/cds_migrator_kit/users/load.py index 736c8a52..2dc5aa05 100644 --- a/cds_migrator_kit/users/load.py +++ b/cds_migrator_kit/users/load.py @@ -46,6 +46,7 @@ def _load(self, entry): """Load users.""" self._owner(entry) self._reviewers(entry) + self._access_grant_emails(entry) def _validate(self, entry): """Validate data before loading.""" @@ -74,6 +75,18 @@ def _reviewers(self, json_entry): else: self._find_or_create_reviewer_by_name(reviewer) + def _access_grant_emails(self, json_entry): + """Fetch or create accounts for direct emails in access grants. + + 506 access restriction fields can name a person directly by email + (see the `access_grants` rule in + cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py) + instead of an e-group/role name - those need an account too, so the + record's actual access grant can later be resolved to a User. + """ + for email in json_entry.get("access_grant_emails", []): + self._find_or_create_by_email(email) + def _find_or_create_by_email(self, email): """Fetch or create a user account by email.""" if not email: diff --git a/cds_migrator_kit/users/transform.py b/cds_migrator_kit/users/transform.py index ec38a06d..4e501fb0 100644 --- a/cds_migrator_kit/users/transform.py +++ b/cds_migrator_kit/users/transform.py @@ -38,7 +38,12 @@ def _transform(self, entry): timestamp, json_data = record_dump.latest_revision email = json_data.get("submitter") reviewers = json_data.get("reviewers", []) - return {"submitter": email, "reviewers": reviewers} + access_grant_emails = json_data.get("access_grant_emails", []) + return { + "submitter": email, + "reviewers": reviewers, + "access_grant_emails": access_grant_emails, + } except Exception as e: cli_logger.exception(e) diff --git a/tests/cds-rdm/test_access_grant_emails.py b/tests/cds-rdm/test_access_grant_emails.py new file mode 100644 index 00000000..6142ad31 --- /dev/null +++ b/tests/cds-rdm/test_access_grant_emails.py @@ -0,0 +1,99 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for pre-creating accounts for direct emails found in field 506. + +See cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py +and cds_migrator_kit/users/load.py::CDSSubmitterLoad._access_grant_emails. +""" + +from cds_dojson.marc21.utils import create_record +from invenio_accounts.testutils import create_test_user + +from cds_migrator_kit.rdm.users.transform.xml_processing.models.submitter import ( + submitter_model, +) +from cds_migrator_kit.users.load import CDSSubmitterLoad + + +def _do(marcxml): + return submitter_model.do(create_record(marcxml)) + + +class TestAccessGrantEmailsRule: + """Test the "^506[1_]_" -> "access_grant_emails" dojson rule.""" + + def test_extracts_email_from_subfield_d(self, base_app): + """A direct email in 506__d (e.g. record 2045640) is picked up.""" + with base_app.app_context(): + out = _do( + """ + + + cds-edboard-dirac@cern.ch + + + """ + ) + assert out["access_grant_emails"] == ["cds-edboard-dirac@cern.ch"] + + def test_ignores_egroup_names(self, base_app): + """E-group names (subfield m/a, no "@") are not treated as emails.""" + with base_app.app_context(): + out = _do( + """ + + + cds-edboard-dirac [CERN] + + + cds-ph-ep-publications-referee-non-lhc [CERN] + + + """ + ) + assert out.get("access_grant_emails", []) == [] + + def test_deduplicates_and_lowercases(self, base_app): + """Repeated/differently-cased emails across occurrences collapse to one.""" + with base_app.app_context(): + out = _do( + """ + + + Jane.Doe@cern.ch + + + jane.doe@cern.ch + + + """ + ) + assert out["access_grant_emails"] == ["jane.doe@cern.ch"] + + +class TestAccessGrantEmailsLoad: + """Test CDSSubmitterLoad._access_grant_emails().""" + + def test_finds_existing_account(self, app, db): + """An email matching an existing account is resolved, not recreated.""" + user = create_test_user(email="cds-edboard-dirac@cern.ch") + db.session.commit() + + load = CDSSubmitterLoad(dry_run=True) + load._access_grant_emails( + {"access_grant_emails": ["cds-edboard-dirac@cern.ch"]} + ) + + found = load._find_or_create_by_email("cds-edboard-dirac@cern.ch") + assert found == user.id + + def test_no_emails_is_a_noop(self, app, db): + """No access_grant_emails key/empty list does nothing.""" + load = CDSSubmitterLoad(dry_run=True) + load._access_grant_emails({}) + load._access_grant_emails({"access_grant_emails": []}) diff --git a/tests/cds-rdm/test_metadata_mappers.py b/tests/cds-rdm/test_metadata_mappers.py new file mode 100644 index 00000000..223707e1 --- /dev/null +++ b/tests/cds-rdm/test_metadata_mappers.py @@ -0,0 +1,75 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for cds_migrator_kit/rdm/records/transform/mappers/metadata.py.""" + +from cds_migrator_kit.rdm.records.transform.mappers.base import ( + RecordTransformContext, +) +from cds_migrator_kit.rdm.records.transform.mappers.metadata import ( + TableOfContentsMapper, +) + + +def _ctx(dojson_entry): + return RecordTransformContext(dojson_entry=dojson_entry, raw_dump_entry={}) + + +class TestTableOfContentsMapper: + """Test TableOfContentsMapper.map_value().""" + + def test_deduplicates_identical_entries(self): + """Exact-duplicate descriptions (e.g. a MARC field repeated in the + legacy record) collapse to a single entry.""" + desc = {"description": "Same abstract text.", "type": {"id": "other"}} + dojson_entry = {"additional_descriptions": [desc, dict(desc), dict(desc)]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc] + + def test_keeps_distinct_entries(self): + """Descriptions that actually differ are all kept, in order.""" + desc_a = {"description": "Series info", "type": {"id": "series-information"}} + desc_b = {"description": "Other note", "type": {"id": "other"}} + dojson_entry = {"additional_descriptions": [desc_a, desc_b]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc_a, desc_b] + + def test_same_text_different_type_is_not_deduplicated(self): + """Same text under a different type is a distinct entry.""" + desc_a = {"description": "Same text", "type": {"id": "other"}} + desc_b = {"description": "Same text", "type": {"id": "series-information"}} + dojson_entry = {"additional_descriptions": [desc_a, desc_b]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc_a, desc_b] + + def test_folds_table_of_content_in_before_deduplicating(self): + """table_of_content is folded in, and still deduped against.""" + toc_entry = { + "description": "1. Intro\n2. Results", + "type": {"id": "table-of-contents"}, + } + dojson_entry = { + "table_of_content": "1. Intro\n2. Results", + "additional_descriptions": [dict(toc_entry)], + } + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [toc_entry] + assert "table_of_content" not in dojson_entry + + def test_no_descriptions_returns_falsy(self): + """No additional_descriptions/table_of_content at all is a no-op.""" + result = TableOfContentsMapper().map_value(_ctx({})) + + assert not result