From caeded845bd51d719afda2a2d0f8728205251ea4 Mon Sep 17 00:00:00 2001 From: kpsherva Date: Wed, 23 Sep 2026 18:10:10 +0200 Subject: [PATCH 1/6] change(config): add dirac and experiment mapping # Conflicts: # cds_migrator_kit/rdm/migration_config.py --- cds_migrator_kit/rdm/migration_config.py | 15 ++++++++++- .../rdm/records/transform/config.py | 8 +++++- cds_migrator_kit/rdm/streams.yaml | 25 +---------------- cds_migrator_kit/rdm/streams_shelved.yaml | 27 ++++++++++++++----- 4 files changed, 42 insertions(+), 33 deletions(-) diff --git a/cds_migrator_kit/rdm/migration_config.py b/cds_migrator_kit/rdm/migration_config.py index 727045bf..ed92679f 100644 --- a/cds_migrator_kit/rdm/migration_config.py +++ b/cds_migrator_kit/rdm/migration_config.py @@ -535,7 +535,10 @@ def resolve_record_pid(pid): ### EP Approval configuration only needed for local, it should use cds-rdm config for de/sandbox/prod # =========================== -CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "78b3c4aa-c4e6-4502-8226-67ba2d347afe" +# ATTENTION: please don't modify this local value - the community is created +# via cds-rdm fixtures with this id - if you have another ID locally +# change it in your local db +CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" """The id of the CERN Scientific community. This is only a local-dev default: on other instances (sandbox/prod), set the @@ -626,4 +629,14 @@ def resolve_record_pid(pid): "counter_digits": 3, # zero-padding width, e.g. 3 → "001" }, }, + "6a289642-5378-4daf-87b5-bb58af00487a": { + # DIRAC + "label": "EP approval", # shown in UI buttons/headings + "referee_group": "cds-ph-ep-publications-referee-non-lhc", # CERN e-group slug + "report_number": { + "prefix": "CERN-EP", # literal prefix, e.g. "CERN-EP" + "include_year": True, # append the current year after prefix + "counter_digits": 3, # zero-padding width, e.g. 3 → "001" + }, + }, } diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 2bca4a47..38e30519 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -161,5 +161,11 @@ # Legacy experiment names remapped to vocabulary ids before lookup EXPERIMENT_ALIASES = { - "t2k": "re13", + "t2k": "RE13", + "antares": "RE6", + "dirac": "PS212", + "dirac ps212": "PS212", + "harp ps214": "PS214", + "harp": "PS214", + "dampe": "RE29", } diff --git a/cds_migrator_kit/rdm/streams.yaml b/cds_migrator_kit/rdm/streams.yaml index 5199d992..933abe14 100644 --- a/cds_migrator_kit/rdm/streams.yaml +++ b/cds_migrator_kit/rdm/streams.yaml @@ -237,33 +237,10 @@ records: communities_ids: - "88a5cdf4-974a-44d4-b145-d19ee6346bb8" - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" - lcd_restr: - plots: true - data_dir: cds_migrator_kit/rdm/data/former_exp/lcd_restr - restricted: "True" - create_inclusion_request: true - extract: - dirpath: cds_migrator_kit/rdm/data/former_exp/lcd_restr - transform: - files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ - missing_users: cds_migrator_kit/rdm/data/users - communities_ids: - - "adc02716-780b-4bba-8e89-6316c9a11cf0" - lcd: - plots: true - create_inclusion_request: true - data_dir: cds_migrator_kit/rdm/data/former_exp/lcd - extract: - dirpath: cds_migrator_kit/rdm/data/former_exp/lcd - transform: - files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ - missing_users: cds_migrator_kit/rdm/data/users - communities_ids: - - "adc02716-780b-4bba-8e89-6316c9a11cf0" - - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" re29: data_dir: cds_migrator_kit/rdm/data/former_exp/re29 plots: true + create_inclusion_request: true extract: dirpath: cds_migrator_kit/rdm/data/former_exp/re29 transform: diff --git a/cds_migrator_kit/rdm/streams_shelved.yaml b/cds_migrator_kit/rdm/streams_shelved.yaml index bd155cc1..371e50a6 100644 --- a/cds_migrator_kit/rdm/streams_shelved.yaml +++ b/cds_migrator_kit/rdm/streams_shelved.yaml @@ -1,16 +1,29 @@ db_uri: postgresql://cds-rdm:cds-rdm@localhost:5432/cds-rdm records: - thesis: - data_dir: cds_migrator_kit/rdm/data/thesis + lcd_restr: + plots: true + data_dir: cds_migrator_kit/rdm/data/former_exp/lcd_restr + restricted: "True" + create_inclusion_request: true extract: - dirpath: cds_migrator_kit/rdm/data/thesis/dump/ + dirpath: cds_migrator_kit/rdm/data/former_exp/lcd_restr transform: - files_dump_dir: cds_migrator_kit/rdm/data/thesis/files/ + files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f - load: - legacy_pids_to_redirect: cds_migrator_kit/rdm/data/thesis/duplicated_pids.json + - "adc02716-780b-4bba-8e89-6316c9a11cf0" + lcd: + plots: true + create_inclusion_request: true + data_dir: cds_migrator_kit/rdm/data/former_exp/lcd + extract: + dirpath: cds_migrator_kit/rdm/data/former_exp/lcd + transform: + files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ + missing_users: cds_migrator_kit/rdm/data/users + communities_ids: + - "adc02716-780b-4bba-8e89-6316c9a11cf0" + - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" mous: data_dir: cds_migrator_kit/rdm/data/mous extract: From a40eedffe14cbe74e557b62ac6af42d0c87bb2c4 Mon Sep 17 00:00:00 2001 From: kpsherva Date: Wed, 23 Sep 2026 18:11:00 +0200 Subject: [PATCH 2/6] change(users): add resolving accounts stored in 506 --- cds_migration_progress.html | 862 ++++++++++++++++++ .../xml_processing/models/submitter.py | 8 +- .../xml_processing/rules/access_grants.py | 54 ++ cds_migrator_kit/users/load.py | 13 + cds_migrator_kit/users/transform.py | 7 +- tests/cds-rdm/test_access_grant_emails.py | 93 ++ 6 files changed, 1035 insertions(+), 2 deletions(-) create mode 100644 cds_migration_progress.html create mode 100644 cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py create mode 100644 tests/cds-rdm/test_access_grant_emails.py diff --git a/cds_migration_progress.html b/cds_migration_progress.html new file mode 100644 index 00000000..e9046ed8 --- /dev/null +++ b/cds_migration_progress.html @@ -0,0 +1,862 @@ +CDS Migration Progress, 2015–2026 + + +
+
+
+

CDS migration progress, 2015–2026

+

Legacy CDS (cds.cern.ch) total record count (estimated including restricted records), plotted on the same scale as cumulative records migrated into repository.cern, cumulative books held in catalogue.library.cern since its 21 April 2021 launch, and cumulative videos held on videos.cern.ch since the 2017 migration.

+
+ +
+
+ Legacy CDS total records (est. incl. restricted) + repository.cern cumulative records + catalogue.library.cern cumulative records + videos.cern.ch cumulative records + Migration milestone +
+ +
+ +
+ +
+ View underlying data (Wayback Machine snapshots & migration events) + + + + +
Legacy CDS (cds.cern.ch) — Wayback Machine snapshots
Snapshot datePublic count (scraped)Retroactive adjustmentEst. total incl. restricted (plotted)repository.cern migrated to dateNet of repository.cern
+ + + + +
repository.cern — migration events
DateContent migratedRecords addedCumulative total
+ + + + +
catalogue.library.cern — books created per year (via /api/documents)
PeriodBooks createdCumulative total
+ + + + +
videos.cern.ch — videos created per year (via /api/records)
PeriodVideos createdCumulative total
+
+ +
+ Legacy CDS source: record counts scraped from the CDS homepage as captured by the Internet Archive Wayback + Machine (web.archive.org, cds.cern.ch), which reflects only publicly visible records. The + confirmed current total including restricted records is 592,089 (2 August 2026); comparing that to the + closest scraped snapshot (566,063, 18 July 2026) implies roughly 26,026 restricted records are not shown by + the public counter. Since we have no historical breakdown of restricted records, that offset is extrapolated + as a constant and added to every earlier scraped snapshot to estimate a total-including-restricted series + (shown in the data table); this assumes the restricted share has stayed roughly flat in absolute terms, which + is a simplification. The final point uses the confirmed live total (592,089, 2 August 2026) directly as its + total, rather than a scraped-plus-offset estimate. Every other plotted point carries a + retroactive adjustment, summing three components, so the total line depicts the drop each + repository.cern migration should produce even though the scraped public counter itself never dips at those + dates (that content isn't purged from legacy CDS on migration). Component A covers CERN Bulletin & + Courier, IT department, and HR content (Nov 2025 – Jan 2026): every point from 2015 through Jan 2024 + carries the same flat allowance (~111,000–114,000, an approximate reconstruction of that content's + size) since none of it had migrated yet; it tapers to +63,000 after CERN Bulletin and Courier moved (Nov + 2025), +3,000 after IT department content moved (Dec 2025), and 0 once HR content had also moved (Jan + 2026). Component B covers Staff Association content and the ALEPH/DELPHI/L3/OPAL research records (Jul + 2026): a flat +13,000 on every point up to 29 July 2026, then 0. Component C covers theses (12,000): a flat + +12,000 on every point up to the 15 May 2025 migration, then 0. The three sum, so points before Nov 2024 + carry all three, points from Nov 2024 through early May 2025 carry A and C, and points from late May 2025 + through Jan 2026 carry only B (or A tapering alongside it from Nov 2025). Three points (Nov & Dec 2024, + Oct 2025) have no underlying wayback snapshot and are linearly interpolated between the nearest real ones + before any adjustment is applied; the data table below marks every reconstructed or boosted point. A + net-of-repository.cern figure — the total minus the cumulative amount migrated to + repository.cern by that date — is also computed and shown in the hover tooltip and the data table, for + readers who want legacy CDS's content net of what's already been migrated out. Neither retroactive + adjustment nor the net-of-repository.cern subtraction applies to the videos.cern.ch or catalogue.library.cern + migrations, since that content did not move to repository.cern. + repository.cern source: cumulative total (130,000) built from known migration batch sizes at each event date + (no continuous archive exists for this newer platform), including a ±1,500 reconciliation against the + confirmed live total to account for imprecise per-event estimates. catalogue.library.cern source: current + total holdings (211,323) queried live from the public /api/documents endpoint; yearly + books-created counts from the same endpoint's _created field, from 21 April 2021 onward, account + for 84,508 of that total, with the remaining 126,815 reconstructed as the baseline migrated on launch day (its + original bibliographic creation dates are preserved from the legacy system, so it isn't separately + timestamped as "migrated"). videos.cern.ch source: current total holdings (39,000, including restricted) + queried live; yearly videos-created counts from the public /api/records endpoint's + _created field, from 2017 onward, account for 17,357 of that total, with the remaining 21,643 + reconstructed as the baseline migrated during the 2017 bulk migration from legacy CDS. Both the + catalogue.library.cern and videos.cern.ch lines therefore step from 0 to their reconstructed baseline at + launch and grow to their live totals from there. All four series share a single record-count scale. Data + compiled 2 August 2026. +
+
+
+ + diff --git a/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py b/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py index a10886e2..c02cb438 100644 --- a/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py +++ b/cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py @@ -146,7 +146,7 @@ class SubmitterModel(CdsOverdo): "542__u", # https://cds.cern.ch/record/2285212/export/hm?ln=en "560172", # https://cds.cern.ch/record/383486/export/hm?ln=en "56017a", # https://cds.cern.ch/record/383486/export/hm?ln=en wrong keyword subfield - "506__m", # mail + # "506__m", # e-group/reader's email, used to find/recreate their account "590__b", # abstract translation "590__a", # abstract translation TODO https://cds.cern.ch/record/1476067/export/hm?ln=en "594__a", # https://cds.cern.ch/record/466504/export/hm?ln=en, 455788 @@ -366,3 +366,9 @@ class SubmitterModel(CdsOverdo): bases=(base_model,), entry_point_group="cds_migrator_kit.migrator.rules.submitter", ) + +# Registers the 506 access-grant-emails rule directly on submitter_model, +# after it has been built above - see access_grants.py's module docstring +# for why this must be isolated to this instance instead of the shared +# base_model. +import cds_migrator_kit.rdm.users.transform.xml_processing.rules.access_grants # noqa: E402,F401 diff --git a/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py b/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py new file mode 100644 index 00000000..db87e021 --- /dev/null +++ b/cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py @@ -0,0 +1,54 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""CDS-RDM access grant accounts migration rules. + +Mirrors the 859__f "submitter" / 906__m "reviewer" rules (see +cds_migrator_kit/transform/xml_processing/rules/base.py and +cds_migrator_kit/rdm/users/transform/xml_processing/rules/reviewers.py): +506 access restriction fields (see the `access_grants` rule in +cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py) +can name a person directly by email, in subfields d/m/a, instead of an +e-group/role name. Those emails need an account pre-created too, so the +record's actual access grant can later be resolved to a `User` +(cds_migrator_kit/rdm/records/transform/entities/parent.py, `resolve_grants`). + +Registered directly on `submitter_model`, not on the shared `base_model`: +research.py/hr.py/it.py/faser_publication.py already register their own, +unrelated "^506[1_]_" rule on their own separate model instances, so this +rule must stay isolated to `submitter_model` to avoid clashing with those. +This is why it is imported at the bottom of +cds_migrator_kit/rdm/users/transform/xml_processing/models/submitter.py, +after `submitter_model` has been constructed. +""" + +import re + +from dojson.errors import IgnoreKey + +from cds_migrator_kit.rdm.users.transform.xml_processing.models.submitter import ( + submitter_model, +) + +EMAIL_PATTERN = re.compile(r"[^@]+@[^@]+\.[^@]+") + + +@submitter_model.over("access_grant_emails", "^506[1_]_") +def record_access_grant_emails(self, key, value): + """Translate 506 access grant emails, ignoring e-group/role names.""" + emails = self.get("access_grant_emails", []) + for subfield in ("d", "m", "a"): + raw = value.get(subfield) + if isinstance(raw, tuple): + raw = raw[0] + if not raw: + continue + candidate = raw.strip().lower() + if EMAIL_PATTERN.match(candidate) and candidate not in emails: + emails.append(candidate) + self["access_grant_emails"] = emails + raise IgnoreKey("access_grant_emails") diff --git a/cds_migrator_kit/users/load.py b/cds_migrator_kit/users/load.py index 736c8a52..2dc5aa05 100644 --- a/cds_migrator_kit/users/load.py +++ b/cds_migrator_kit/users/load.py @@ -46,6 +46,7 @@ def _load(self, entry): """Load users.""" self._owner(entry) self._reviewers(entry) + self._access_grant_emails(entry) def _validate(self, entry): """Validate data before loading.""" @@ -74,6 +75,18 @@ def _reviewers(self, json_entry): else: self._find_or_create_reviewer_by_name(reviewer) + def _access_grant_emails(self, json_entry): + """Fetch or create accounts for direct emails in access grants. + + 506 access restriction fields can name a person directly by email + (see the `access_grants` rule in + cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py) + instead of an e-group/role name - those need an account too, so the + record's actual access grant can later be resolved to a User. + """ + for email in json_entry.get("access_grant_emails", []): + self._find_or_create_by_email(email) + def _find_or_create_by_email(self, email): """Fetch or create a user account by email.""" if not email: diff --git a/cds_migrator_kit/users/transform.py b/cds_migrator_kit/users/transform.py index ec38a06d..4e501fb0 100644 --- a/cds_migrator_kit/users/transform.py +++ b/cds_migrator_kit/users/transform.py @@ -38,7 +38,12 @@ def _transform(self, entry): timestamp, json_data = record_dump.latest_revision email = json_data.get("submitter") reviewers = json_data.get("reviewers", []) - return {"submitter": email, "reviewers": reviewers} + access_grant_emails = json_data.get("access_grant_emails", []) + return { + "submitter": email, + "reviewers": reviewers, + "access_grant_emails": access_grant_emails, + } except Exception as e: cli_logger.exception(e) diff --git a/tests/cds-rdm/test_access_grant_emails.py b/tests/cds-rdm/test_access_grant_emails.py new file mode 100644 index 00000000..583a1a99 --- /dev/null +++ b/tests/cds-rdm/test_access_grant_emails.py @@ -0,0 +1,93 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for pre-creating accounts for direct emails found in field 506. + +See cds_migrator_kit/rdm/users/transform/xml_processing/rules/access_grants.py +and cds_migrator_kit/users/load.py::CDSSubmitterLoad._access_grant_emails. +""" + +from cds_dojson.marc21.utils import create_record +from invenio_accounts.testutils import create_test_user + +from cds_migrator_kit.rdm.users.transform.xml_processing.models.submitter import ( + submitter_model, +) +from cds_migrator_kit.users.load import CDSSubmitterLoad + + +def _do(marcxml): + return submitter_model.do(create_record(marcxml)) + + +class TestAccessGrantEmailsRule: + """Test the "^506[1_]_" -> "access_grant_emails" dojson rule.""" + + def test_extracts_email_from_subfield_d(self, base_app): + """A direct email in 506__d (e.g. record 2045640) is picked up.""" + with base_app.app_context(): + out = _do(""" + + + cds-edboard-dirac@cern.ch + + + """) + assert out["access_grant_emails"] == ["cds-edboard-dirac@cern.ch"] + + def test_ignores_egroup_names(self, base_app): + """E-group names (subfield m/a, no "@") are not treated as emails.""" + with base_app.app_context(): + out = _do(""" + + + cds-edboard-dirac [CERN] + + + cds-ph-ep-publications-referee-non-lhc [CERN] + + + """) + assert out.get("access_grant_emails", []) == [] + + def test_deduplicates_and_lowercases(self, base_app): + """Repeated/differently-cased emails across occurrences collapse to one.""" + with base_app.app_context(): + out = _do(""" + + + Jane.Doe@cern.ch + + + jane.doe@cern.ch + + + """) + assert out["access_grant_emails"] == ["jane.doe@cern.ch"] + + +class TestAccessGrantEmailsLoad: + """Test CDSSubmitterLoad._access_grant_emails().""" + + def test_finds_existing_account(self, app, db): + """An email matching an existing account is resolved, not recreated.""" + user = create_test_user(email="cds-edboard-dirac@cern.ch") + db.session.commit() + + load = CDSSubmitterLoad(dry_run=True) + load._access_grant_emails( + {"access_grant_emails": ["cds-edboard-dirac@cern.ch"]} + ) + + found = load._find_or_create_by_email("cds-edboard-dirac@cern.ch") + assert found == user.id + + def test_no_emails_is_a_noop(self, app, db): + """No access_grant_emails key/empty list does nothing.""" + load = CDSSubmitterLoad(dry_run=True) + load._access_grant_emails({}) + load._access_grant_emails({"access_grant_emails": []}) From 93d3c9b200c5d49203d1a2a148cb721f69df9869 Mon Sep 17 00:00:00 2001 From: kpsherva Date: Wed, 23 Sep 2026 18:12:29 +0200 Subject: [PATCH 3/6] change(transform): improve the metadata mapping, deduplicate descriptions --- .../transform/mappers/custom_fields.py | 2 +- .../rdm/records/transform/mappers/metadata.py | 23 +++++- .../rdm/records/transform/models/research.py | 2 +- .../transform/models/research_committee.py | 3 + .../transform/xml_processing/rules/base.py | 2 +- .../xml_processing/rules/research.py | 3 + tests/cds-rdm/test_metadata_mappers.py | 75 +++++++++++++++++++ 7 files changed, 105 insertions(+), 5 deletions(-) create mode 100644 tests/cds-rdm/test_metadata_mappers.py diff --git a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py index 7150026b..4b8736c6 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py @@ -153,7 +153,7 @@ def apply(self, ctx): "cern:accelerators", [] ) for accelerator in accelerators: - if accelerator.lower().strip() in ["not applicable", "xx", "fermi"]: + if accelerator.lower().strip() in ["not applicable", "xx", "fermi", "cern recognized expt."]: continue result = search_vocabulary(accelerator, "accelerators") if result and result not in accelerators_out: diff --git a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py index f23f9f5d..d3135d27 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py @@ -189,12 +189,23 @@ def map_value(self, ctx): class TableOfContentsMapper(FieldMapper): - """Folds table_of_content into additional_descriptions.""" + """Folds table_of_content into additional_descriptions. + + Also the single place where the final ``additional_descriptions`` list + is deduplicated: many different dojson rules append to it (520/246/ + 035/500/210/... across base.py and the various collection-specific + rule modules), some legacy records repeat the very same MARC field + (identical text, sometimes only differing in a provenance subfield + nothing here reads), and not every one of those rules remembers to + guard against re-adding an entry already present. Deduplicating once + here, after every rule has run, doesn't depend on each of them getting + that guard right. + """ id = "additional_descriptions" def map_value(self, ctx): - """Move table_of_content into additional_descriptions and return it.""" + """Move table_of_content into additional_descriptions and dedupe.""" dojson_entry = ctx.dojson_entry toc = dojson_entry.get("table_of_content", []) additional_desc = dojson_entry.get("additional_descriptions", []) @@ -204,6 +215,14 @@ def map_value(self, ctx): ) dojson_entry["additional_descriptions"] = additional_desc dojson_entry.pop("table_of_content") + + deduped = [] + for description in dojson_entry.get("additional_descriptions", []): + if description not in deduped: + deduped.append(description) + if deduped: + dojson_entry["additional_descriptions"] = deduped + return dojson_entry.get("additional_descriptions") diff --git a/cds_migrator_kit/rdm/records/transform/models/research.py b/cds_migrator_kit/rdm/records/transform/models/research.py index 4b8e0b78..701b6f36 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research.py +++ b/cds_migrator_kit/rdm/records/transform/models/research.py @@ -16,7 +16,7 @@ class ResearchModel(CdsOverdo): """Translation model for research.""" - __query__ = '693__.e:"DAMPE RE29" OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM' + __query__ = '693__.e:"DAMPE RE29" OR 693__.e:RE29 OR 693__.e:DAMPE OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM -980:BULLETINNEWS' __ignore_keys__ = { "0248_a", diff --git a/cds_migrator_kit/rdm/records/transform/models/research_committee.py b/cds_migrator_kit/rdm/records/transform/models/research_committee.py index 3135e64b..13d08b02 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research_committee.py +++ b/cds_migrator_kit/rdm/records/transform/models/research_committee.py @@ -42,12 +42,15 @@ class ResearchCommitteeModel(CdsOverdo): "340__a", # TODO ignore material? "540__3", # TODO still ignore the material of the license? "542__3", # TODO still ignore the material of the license? + "594__a", # ATN tag "595__i", # TODO ?? "695__e", # some inspire tag + "695__9", # some inspire tag "700__m", # email of contributor "700__q", # TODO ignore? aliteration of the name, used for searching "700__v", # TODO drop? "773__x", # INSPIRE publication note + "852__a", "8564_8", # file id "8564_s", # bibdoc id "8564_x", # icon thumbnails sizes diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py index 695a4e3a..4fb25884 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py @@ -924,7 +924,7 @@ def related_identifiers_787(self, key, value): "resource_type": {"id": "publication-report"}, }, "complemented by": { - "relation_type": {"id": "issuplementedby"}, + "relation_type": {"id": "issupplementedby"}, "resource_type": {"id": "publication-report"}, }, "preprint": { diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py index e1ac8b33..c9c444aa 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py @@ -748,6 +748,9 @@ def resource_type(self, key, value): "lhcf_proc": {"id": "publication-conferenceproceeding"}, "lhcf_reports": {"id": "publication-report"}, "conferencepapers": {"id": "publication-conferencepaper"}, + "technical note": {"id": "publication-technicalnote"}, + "minutes": {"id": "publication-meetingminutes"}, + "presentation": {"id": "presentation"}, } try: diff --git a/tests/cds-rdm/test_metadata_mappers.py b/tests/cds-rdm/test_metadata_mappers.py new file mode 100644 index 00000000..223707e1 --- /dev/null +++ b/tests/cds-rdm/test_metadata_mappers.py @@ -0,0 +1,75 @@ +# -*- coding: utf-8 -*- +# +# Copyright (C) 2026 CERN. +# +# CDS-RDM is free software; you can redistribute it and/or modify it under +# the terms of the MIT License; see LICENSE file for more details. + +"""Tests for cds_migrator_kit/rdm/records/transform/mappers/metadata.py.""" + +from cds_migrator_kit.rdm.records.transform.mappers.base import ( + RecordTransformContext, +) +from cds_migrator_kit.rdm.records.transform.mappers.metadata import ( + TableOfContentsMapper, +) + + +def _ctx(dojson_entry): + return RecordTransformContext(dojson_entry=dojson_entry, raw_dump_entry={}) + + +class TestTableOfContentsMapper: + """Test TableOfContentsMapper.map_value().""" + + def test_deduplicates_identical_entries(self): + """Exact-duplicate descriptions (e.g. a MARC field repeated in the + legacy record) collapse to a single entry.""" + desc = {"description": "Same abstract text.", "type": {"id": "other"}} + dojson_entry = {"additional_descriptions": [desc, dict(desc), dict(desc)]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc] + + def test_keeps_distinct_entries(self): + """Descriptions that actually differ are all kept, in order.""" + desc_a = {"description": "Series info", "type": {"id": "series-information"}} + desc_b = {"description": "Other note", "type": {"id": "other"}} + dojson_entry = {"additional_descriptions": [desc_a, desc_b]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc_a, desc_b] + + def test_same_text_different_type_is_not_deduplicated(self): + """Same text under a different type is a distinct entry.""" + desc_a = {"description": "Same text", "type": {"id": "other"}} + desc_b = {"description": "Same text", "type": {"id": "series-information"}} + dojson_entry = {"additional_descriptions": [desc_a, desc_b]} + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [desc_a, desc_b] + + def test_folds_table_of_content_in_before_deduplicating(self): + """table_of_content is folded in, and still deduped against.""" + toc_entry = { + "description": "1. Intro\n2. Results", + "type": {"id": "table-of-contents"}, + } + dojson_entry = { + "table_of_content": "1. Intro\n2. Results", + "additional_descriptions": [dict(toc_entry)], + } + + result = TableOfContentsMapper().map_value(_ctx(dojson_entry)) + + assert result == [toc_entry] + assert "table_of_content" not in dojson_entry + + def test_no_descriptions_returns_falsy(self): + """No additional_descriptions/table_of_content at all is a no-op.""" + result = TableOfContentsMapper().map_value(_ctx({})) + + assert not result From f9bd6726299ad620c192c77afe35feb2d7e0f66b Mon Sep 17 00:00:00 2001 From: kpsherva Date: Fri, 2 Oct 2026 13:35:33 +0200 Subject: [PATCH 4/6] add(mappers): drop broken links, map descriptions --- .../rdm/records/transform/config.py | 4 ++++ .../transform/mappers/custom_fields.py | 11 ++++++++- .../rdm/records/transform/mappers/metadata.py | 24 ++++++++++++++++++- .../rdm/records/transform/mappers/registry.py | 2 ++ .../transform/xml_processing/rules/base.py | 22 +++++++++++++++++ 5 files changed, 61 insertions(+), 2 deletions(-) diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 38e30519..baa8985b 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -169,3 +169,7 @@ "harp": "PS214", "dampe": "RE29", } + +# 693__e values that are keywords rather than real experiments (curated as +# "mots clef"), mapped to subjects instead of cern:experiments +EXPERIMENTS_AS_SUBJECTS = ["d3", "r104", "r105a"] diff --git a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py index 4b8736c6..895d2015 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py @@ -12,7 +12,10 @@ RecordFlaggedCuration, UnexpectedValue, ) -from cds_migrator_kit.rdm.records.transform.config import EXPERIMENT_ALIASES +from cds_migrator_kit.rdm.records.transform.config import ( + EXPERIMENT_ALIASES, + EXPERIMENTS_AS_SUBJECTS, +) from cds_migrator_kit.rdm.records.transform.mappers.base import CustomFieldMapper from cds_migrator_kit.rdm.records.transform.mappers.vocabulary import search_vocabulary @@ -34,6 +37,12 @@ def apply(self, ctx): for experiment in experiments: if experiment.lower().strip() in ["not applicable", "xx"]: continue + if experiment.lower().strip() in EXPERIMENTS_AS_SUBJECTS: + # curated as keywords ("mots clef"), not real experiments + ctx.dojson_entry.setdefault("subjects", []).append( + {"subject": experiment} + ) + continue experiment = EXPERIMENT_ALIASES.get(experiment.lower().strip(), experiment) result = search_vocabulary(experiment, "experiments") if result and result not in experiments_out: diff --git a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py index d3135d27..1c8a7638 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py @@ -263,6 +263,29 @@ def map_value(self, ctx): return identifiers +#: `setlink` is a CDS-internal redirector, not a real related resource - +#: drop any related_identifiers entry pointing at it. +_SETLINK_URL_PREFIX = "http://documents.cern.ch/cgi-bin/setlink?" + + +class RelatedIdentifiersMapper(FieldMapper): + """Maps related_identifiers, dropping CDS-internal setlink URLs.""" + + id = "related_identifiers" + + def map_value(self, ctx): + """Return related_identifiers with setlink URL entries removed.""" + related_identifiers = ctx.dojson_entry.get("related_identifiers", []) + return [ + item + for item in related_identifiers + if not ( + (item.get("scheme") or "").upper() == "URL" + and (item.get("identifier") or "").startswith(_SETLINK_URL_PREFIX) + ) + ] + + # Fields that pass through unchanged from dojson_entry - kept explicit in the # composed list (mappers/config equivalent) rather than open-ended, so the # "forgotten metadata key" completeness check in @@ -275,7 +298,6 @@ def map_value(self, ctx): "languages", "dates", "funding", - "related_identifiers", "rights", "copyright", ) diff --git a/cds_migrator_kit/rdm/records/transform/mappers/registry.py b/cds_migrator_kit/rdm/records/transform/mappers/registry.py index abe28f11..938d1cb7 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/registry.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/registry.py @@ -27,6 +27,7 @@ PASSTHROUGH_METADATA_FIELDS, IdentifiersMapper, PublicationDateMapper, + RelatedIdentifiersMapper, ResourceTypeMapper, SubjectsMapper, TableOfContentsMapper, @@ -45,6 +46,7 @@ PublicationDateMapper(), SubjectsMapper(), IdentifiersMapper(), + RelatedIdentifiersMapper(), *(PassthroughMapper(field_name) for field_name in PASSTHROUGH_METADATA_FIELDS), ) diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py index 4fb25884..104aed23 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py @@ -333,6 +333,9 @@ def report_number(self, key, value): raise IgnoreKey("related_identifiers") elif scheme.upper().startswith("B00"): raise IgnoreKey("related_identifiers") + elif key == "088__" and scheme.upper().startswith("SC000"): + # internal scanning/digitisation request number, to drop + raise IgnoreKey("related_identifiers") elif scheme.startswith("SCOO"): identifier = scheme scheme = "other" @@ -828,6 +831,20 @@ def series_information(self, key, value): return {"description": series, "type": {"id": "series-information"}} +@model.over("additional_descriptions", "^336__") +@for_each_value +def multiple_videos_note(self, key, value): + """Translate the video-system's "multiple videos" cross-reference note. + + The only recognised 336__a content - any other value is unexpected and + flagged for manual curation rather than silently dropped. + """ + note = StringValue(value.get("a", "")).parse() + if not note.startswith("Multiple videos have been identified with recid"): + raise UnexpectedValue(field=key, subfield="a", value=value) + return {"description": note, "type": {"id": "technical-info"}} + + @model.over("related_identifiers", "^084__") @for_each_value def yellow_reports(self, key, value): @@ -931,6 +948,11 @@ def related_identifiers_787(self, key, value): "relation_type": {"id": "references"}, "resource_type": {"id": "publication-preprint"}, }, + "related video": { + "relation_type": {"id": "references"}, + "resource_type": {"id": "video"}, + }, + } if recid: From 00a217d9d6b9b5e25741f9e3f62d5936bdde986c Mon Sep 17 00:00:00 2001 From: kpsherva Date: Fri, 2 Oct 2026 13:36:25 +0200 Subject: [PATCH 5/6] change(chore): formatting --- cds_migrator_kit/rdm/migration_config.py | 16 ++++++++-------- cds_migrator_kit/rdm/records/load/load.py | 18 +++++++++++++++++- .../records/transform/mappers/custom_fields.py | 7 ++++++- .../rdm/records/transform/models/research.py | 1 + .../transform/xml_processing/rules/base.py | 1 - cds_migrator_kit/rdm/streams.yaml | 5 +++-- tests/cds-rdm/test_access_grant_emails.py | 18 ++++++++++++------ 7 files changed, 47 insertions(+), 19 deletions(-) diff --git a/cds_migrator_kit/rdm/migration_config.py b/cds_migrator_kit/rdm/migration_config.py index ed92679f..29ca7c73 100644 --- a/cds_migrator_kit/rdm/migration_config.py +++ b/cds_migrator_kit/rdm/migration_config.py @@ -630,13 +630,13 @@ def resolve_record_pid(pid): }, }, "6a289642-5378-4daf-87b5-bb58af00487a": { - # DIRAC - "label": "EP approval", # shown in UI buttons/headings - "referee_group": "cds-ph-ep-publications-referee-non-lhc", # CERN e-group slug - "report_number": { - "prefix": "CERN-EP", # literal prefix, e.g. "CERN-EP" - "include_year": True, # append the current year after prefix - "counter_digits": 3, # zero-padding width, e.g. 3 → "001" - }, + # DIRAC + "label": "EP approval", # shown in UI buttons/headings + "referee_group": "cds-ph-ep-publications-referee-non-lhc", # CERN e-group slug + "report_number": { + "prefix": "CERN-EP", # literal prefix, e.g. "CERN-EP" + "include_year": True, # append the current year after prefix + "counter_digits": 3, # zero-padding width, e.g. 3 → "001" + }, }, } diff --git a/cds_migrator_kit/rdm/records/load/load.py b/cds_migrator_kit/rdm/records/load/load.py index 4e18d000..86778c88 100644 --- a/cds_migrator_kit/rdm/records/load/load.py +++ b/cds_migrator_kit/rdm/records/load/load.py @@ -22,7 +22,7 @@ from invenio_db import db from invenio_db.uow import ModelCommitOp, UnitOfWork from invenio_i18n import _ -from invenio_pidstore.errors import PIDDoesNotExistError +from invenio_pidstore.errors import PIDAlreadyExists, PIDDoesNotExistError from invenio_pidstore.models import PersistentIdentifier from invenio_rdm_migrator.load.base import Load from invenio_rdm_records.proxies import current_rdm_records_service @@ -308,6 +308,22 @@ def _load(self, entry: MigrationEntry): # apply after record fully finished (does not sync at the spot, only enabled) self._apply_clc_sync(recid_state_after_load, entry) return recid_state_after_load + except PIDAlreadyExists: + # The legacy recid's `lrecid` PID was minted by someone else + # between our _should_skip_recid() check above and now - e.g. a + # concurrently running migration of another collection whose + # dump cross-lists the same legacy recid (two former-experiment + # collections can both ship the same record). Treat it exactly + # like _should_skip_recid: already migrated, nothing to do. + # this happens when you run several mirations at the same time + self.migration_logger.add_information( + recid, + state={ + "message": "Record already migrated (lrecid PID already minted)", + "value": recid, + }, + ) + self.migration_logger.finalise_record(recid) except (UnexpectedValue, ManualImportRequired, GrantCreationError) as e: self.migration_logger.add_log(e, record=entry) except (CDSMigrationException, ValidationError, InvalidRelationValue) as e: diff --git a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py index 895d2015..84a1e56e 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py @@ -162,7 +162,12 @@ def apply(self, ctx): "cern:accelerators", [] ) for accelerator in accelerators: - if accelerator.lower().strip() in ["not applicable", "xx", "fermi", "cern recognized expt."]: + if accelerator.lower().strip() in [ + "not applicable", + "xx", + "fermi", + "cern recognized expt.", + ]: continue result = search_vocabulary(accelerator, "accelerators") if result and result not in accelerators_out: diff --git a/cds_migrator_kit/rdm/records/transform/models/research.py b/cds_migrator_kit/rdm/records/transform/models/research.py index 701b6f36..59e2dd5b 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research.py +++ b/cds_migrator_kit/rdm/records/transform/models/research.py @@ -50,6 +50,7 @@ class ResearchModel(CdsOverdo): "542__8", # agreed not to migrate open access related fields "595__i", # TODO ?? "695__e", # some inspire tag + "695__9", # bibclassify "700__m", # email of contributor "700__q", # TODO ignore? aliteration of the name, used for searching "700__v", # TODO drop? diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py index 104aed23..017a04cb 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py @@ -952,7 +952,6 @@ def related_identifiers_787(self, key, value): "relation_type": {"id": "references"}, "resource_type": {"id": "video"}, }, - } if recid: diff --git a/cds_migrator_kit/rdm/streams.yaml b/cds_migrator_kit/rdm/streams.yaml index 933abe14..c8a9dc2e 100644 --- a/cds_migrator_kit/rdm/streams.yaml +++ b/cds_migrator_kit/rdm/streams.yaml @@ -141,9 +141,9 @@ records: - "28bf99b9-1e72-405a-b95a-828019831def" - "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f" scp_adv: - data_dir: cds_migrator_kit/rdm/data/committees/sc_sdv + data_dir: cds_migrator_kit/rdm/data/committees/scp_sdv extract: - dirpath: cds_migrator_kit/rdm/data/committees/sc_adv + dirpath: cds_migrator_kit/rdm/data/committees/scp_adv transform: files_dump_dir: cds_migrator_kit/rdm/data/committees/files/ missing_users: cds_migrator_kit/rdm/data/users @@ -267,6 +267,7 @@ records: data_dir: cds_migrator_kit/rdm/data/former_exp/ua2 extract: dirpath: cds_migrator_kit/rdm/data/former_exp/ua2 + preferred_model: research transform: files_dump_dir: cds_migrator_kit/rdm/data/former_exp/files/ missing_users: cds_migrator_kit/rdm/data/users diff --git a/tests/cds-rdm/test_access_grant_emails.py b/tests/cds-rdm/test_access_grant_emails.py index 583a1a99..6142ad31 100644 --- a/tests/cds-rdm/test_access_grant_emails.py +++ b/tests/cds-rdm/test_access_grant_emails.py @@ -30,19 +30,22 @@ class TestAccessGrantEmailsRule: def test_extracts_email_from_subfield_d(self, base_app): """A direct email in 506__d (e.g. record 2045640) is picked up.""" with base_app.app_context(): - out = _do(""" + out = _do( + """ cds-edboard-dirac@cern.ch - """) + """ + ) assert out["access_grant_emails"] == ["cds-edboard-dirac@cern.ch"] def test_ignores_egroup_names(self, base_app): """E-group names (subfield m/a, no "@") are not treated as emails.""" with base_app.app_context(): - out = _do(""" + out = _do( + """ cds-edboard-dirac [CERN] @@ -51,13 +54,15 @@ def test_ignores_egroup_names(self, base_app): cds-ph-ep-publications-referee-non-lhc [CERN] - """) + """ + ) assert out.get("access_grant_emails", []) == [] def test_deduplicates_and_lowercases(self, base_app): """Repeated/differently-cased emails across occurrences collapse to one.""" with base_app.app_context(): - out = _do(""" + out = _do( + """ Jane.Doe@cern.ch @@ -66,7 +71,8 @@ def test_deduplicates_and_lowercases(self, base_app): jane.doe@cern.ch - """) + """ + ) assert out["access_grant_emails"] == ["jane.doe@cern.ch"] From 9a7fe36f44cdac1d9b6667b2b58aa3d67203707b Mon Sep 17 00:00:00 2001 From: kpsherva Date: Tue, 6 Oct 2026 18:24:46 +0200 Subject: [PATCH 6/6] change(mappers): don't duplicate DOIs to related_identifiers --- .../rdm/records/transform/entities/record.py | 8 +++++--- cds_migrator_kit/rdm/records/transform/mappers/base.py | 1 + .../rdm/records/transform/mappers/metadata.py | 10 +++++++++- cds_migrator_kit/rdm/records/transform/models/lep.py | 1 + 4 files changed, 16 insertions(+), 4 deletions(-) diff --git a/cds_migrator_kit/rdm/records/transform/entities/record.py b/cds_migrator_kit/rdm/records/transform/entities/record.py index 4a574c91..9e386dd9 100644 --- a/cds_migrator_kit/rdm/records/transform/entities/record.py +++ b/cds_migrator_kit/rdm/records/transform/entities/record.py @@ -168,7 +168,7 @@ def _files(self, dump): files = dump.files return {"enabled": bool(files)} - def _metadata(self, dojson_entry, raw_dump_entry): + def _metadata(self, dojson_entry, raw_dump_entry, pids=None): """Build the metadata dict by running the composed field mappers. Whether every ``dojson_entry`` key ended up consumed *somewhere* in @@ -183,6 +183,7 @@ def _metadata(self, dojson_entry, raw_dump_entry): raw_dump_entry=raw_dump_entry, migration_logger=self.migration_logger, affiliations_mapping=self.affiliations_mapping, + pids=pids or {}, ) metadata = ctx.metadata # Order matters: ResourceTypeMapper must run before TitleMapper reads @@ -259,10 +260,11 @@ def build(self) -> RecordBody: # same reason as _pids()/_access()'s record_restriction pop. internal_notes = dojson_entry.pop("internal_notes", None) + pids = self._pids(dojson_entry) record_json_output = { "files": self._files(dump), - "pids": self._pids(dojson_entry), - "metadata": self._metadata(dojson_entry, raw_dump_entry), + "pids": pids, + "metadata": self._metadata(dojson_entry, raw_dump_entry, pids), "internal_notes": internal_notes, "custom_fields": custom_fields, } diff --git a/cds_migrator_kit/rdm/records/transform/mappers/base.py b/cds_migrator_kit/rdm/records/transform/mappers/base.py index ec811f5f..506679bb 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/base.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/base.py @@ -28,6 +28,7 @@ class RecordTransformContext: migration_logger: object = None affiliations_mapping: object = None access_grants_view: object = None + pids: dict = field(default_factory=dict) metadata: dict = field(default_factory=dict) custom_fields: dict = field(default_factory=dict) diff --git a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py index 1c8a7638..ca00b2b1 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/metadata.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/metadata.py @@ -274,8 +274,10 @@ class RelatedIdentifiersMapper(FieldMapper): id = "related_identifiers" def map_value(self, ctx): - """Return related_identifiers with setlink URL entries removed.""" + """Return related_identifiers without setlink URLs or the record's own DOI.""" related_identifiers = ctx.dojson_entry.get("related_identifiers", []) + record_doi = ((ctx.pids or {}).get("doi") or {}).get("identifier") + record_doi = record_doi.strip().lower() if record_doi else None return [ item for item in related_identifiers @@ -283,6 +285,12 @@ def map_value(self, ctx): (item.get("scheme") or "").upper() == "URL" and (item.get("identifier") or "").startswith(_SETLINK_URL_PREFIX) ) + # the record's own DOI is already in pids, don't repeat it + and not ( + record_doi + and (item.get("scheme") or "").lower() == "doi" + and (item.get("identifier") or "").strip().lower() == record_doi + ) ] diff --git a/cds_migrator_kit/rdm/records/transform/models/lep.py b/cds_migrator_kit/rdm/records/transform/models/lep.py index 53fcadd4..46417bfd 100644 --- a/cds_migrator_kit/rdm/records/transform/models/lep.py +++ b/cds_migrator_kit/rdm/records/transform/models/lep.py @@ -20,6 +20,7 @@ class LEPResearchModel(ResearchModel): __ignore_keys__ = { "594__a", # can be ignored for this collection + "852__a", # location "775__p", # can be ignored for this collection - title of another volume "775__c", # year of volume "596__a", # multivolume tag