diff --git a/cds_migrator_kit/rdm/records/transform/config.py b/cds_migrator_kit/rdm/records/transform/config.py index 2bca4a47..196fd10b 100644 --- a/cds_migrator_kit/rdm/records/transform/config.py +++ b/cds_migrator_kit/rdm/records/transform/config.py @@ -66,7 +66,7 @@ "aip", "jacow", ] -KEYWORD_SCHEMES_TO_DROP = ["proquest", "disxa", "inspeq"] +KEYWORD_SCHEMES_TO_DROP = ["proquest", "disxa", "inspeq", "jinr"] ALLOWED_THESIS_COLLECTIONS = [ "thesis", @@ -162,4 +162,6 @@ # Legacy experiment names remapped to vocabulary ids before lookup EXPERIMENT_ALIASES = { "t2k": "re13", + "compass": "NA58", + "compass na58": "NA58", } diff --git a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py index 7150026b..20f2e5a5 100644 --- a/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py +++ b/cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py @@ -232,6 +232,9 @@ class JournalMapper(CustomFieldMapper): def apply(self, ctx): """Set ctx.custom_fields["journal:journal"].""" + # `_773_m_seen` is an internal bookkeeping field, just to check that we only have one 773__m + # per record as agreed with SIS. + ctx.dojson_entry.pop("_773_m_seen", None) journal = ctx.dojson_entry.get("custom_fields", {}).get("journal:journal", {}) if journal and not journal.get("title"): ctx.flag_curation( diff --git a/cds_migrator_kit/rdm/records/transform/models/_config.py b/cds_migrator_kit/rdm/records/transform/models/_config.py index 15b3cf54..f047f122 100644 --- a/cds_migrator_kit/rdm/records/transform/models/_config.py +++ b/cds_migrator_kit/rdm/records/transform/models/_config.py @@ -7,27 +7,28 @@ "0248_q", "852__c", # holdings will be taken separately "852__h", - # "035__h", # OAI harvest tag or timestamp - # "035__d", # OAI harvest tag or timestamp - # "035__m", # OAI harvest format (e.g. `marcxml`) - # "035__t", # oai harvest tag - # "035__u", # oai harvest tag - # "035__z", # oai harvest tag + "035__h", # OAI harvest tag or timestamp + "035__d", # OAI harvest tag or timestamp + "035__m", # OAI harvest format (e.g. `marcxml`) + "035__t", # oai harvest tag + "035__u", # oai harvest tag + "035__z", # oai harvest tag "037__c", # arxiv subject "100__m", # email of contributor "245__9", # Provenance of title - # "270__m", # Contact email "300__a", # number of pages + "340__a", # See decision log "500__9", # Provenance of the note "520__9", # Provenance of the description - # "540__3", # Material of the license - # "540__9", # Also material of the license - # "542__3", # Also material of the license + "540__3", # Material of the license + "540__9", # Also material of the license + "540__g", # From SIS (see decision log) + "542__3", # Also material of the license "700__m", # email of contributor - # "773__t", # from SIS: can be ignored - # "773__0", # from SIS: can be ignored - # "773__o", # from SIS: can be ignored - # "773__x", # INSPIRE publication note + "773__t", # from SIS: can be ignored + "773__0", # from SIS: can be ignored + "773__o", # from SIS: can be ignored + "773__x", # INSPIRE publication note "8564_8", # file id "8564_s", # bibdoc id "8564_x", # icon thumbnails sizes @@ -35,6 +36,7 @@ "8564_8", # File information (done by file dump) "8564_q", # File links File information (done by file dump) "8564_z", # Websubmit "stamp" (migrated as file metadata) + "905__m", # Submitter email address "916__y", # year, redundant value "937__c", # last modified by "937__s", # last modification date @@ -68,4 +70,6 @@ "999C6t", # https://cds.cern.ch/record/2284606/export/hm?ln=en "999C6v", # https://cds.cern.ch/record/2284606/export/hm?ln=en "999C5d", # old INSPIRE attr + "999C69", # See decision log + "999C6c", # See decision log } diff --git a/cds_migrator_kit/rdm/records/transform/models/north_area.py b/cds_migrator_kit/rdm/records/transform/models/north_area.py index f4580abf..736e9e9b 100644 --- a/cds_migrator_kit/rdm/records/transform/models/north_area.py +++ b/cds_migrator_kit/rdm/records/transform/models/north_area.py @@ -5,7 +5,7 @@ # CDS-RDM is free software; you can redistribute it and/or modify it under # the terms of the MIT License; see LICENSE file for more details. -"""CDS-RDM North Area models (NA61-64).""" +"""CDS-RDM North Area models (NA58 & NA61-66).""" from cds_migrator_kit.rdm.records.transform.models._config import IGNORE_SYSTEM_KEYS from cds_migrator_kit.rdm.records.transform.models.base_publication_record import ( @@ -17,14 +17,19 @@ class NorthAreaModel(CdsOverdo): """Translation model for North Area experiments.""" - __query__ = """693__.e:"NA61" OR 693__.e:"SHINE NA61" OR 693__.e:"NA62" OR 693__.e:"NA63" OR 693__.e:"NA64" OR 693__.e:"DsTau NA65" OR 693__.e:"AMBER NA66" - -980__:THESIS -980__:DELETED -980__:HIDDEN -980__:DUMMY""" + __query__ = """ + 693__.e:"COMPASS" OR 693__.e:"COMPASS NA58" OR 693__.e:"NA61" OR 693__.e:"SHINE NA61" OR + 693__.e:"NA62" OR 693__.e:"NA63" OR 693__.e:"NA64" OR 693__.e:"DsTau NA65" OR 693__.e:"AMBER NA66" + -037__:CERN-STUDENTS-Note-* -690C_:SCICOM -980__:THESIS -980__:DELETED -980__:HIDDEN -980__:DUMMY + """ __ignore_keys__ = IGNORE_SYSTEM_KEYS | { - "270__m", # Email of contact person "500__9", # Provenance of the note + "595_Da", # From SIS: these can be ignored + "595_Dd", # From SIS: these can be ignored + "595_Ds", # From SIS: these can be ignored + "595__9", # From SIS: these can be ignored "903__s", # 'public' - "905__m", # Submitter email address "995__a", # "Inspire" } diff --git a/cds_migrator_kit/rdm/records/transform/models/research.py b/cds_migrator_kit/rdm/records/transform/models/research.py index 994bdc26..754d7ed4 100644 --- a/cds_migrator_kit/rdm/records/transform/models/research.py +++ b/cds_migrator_kit/rdm/records/transform/models/research.py @@ -39,7 +39,6 @@ class ResearchModel(CdsOverdo): "037__c", # arxiv subject "100__m", # email of contributor "245__9", # title provenance - "270__m", # document contact email "300__a", # number of pages "340__a", # TODO ignore material? "540__3", # TODO still ignore the material of the license? diff --git a/cds_migrator_kit/rdm/records/transform/transform.py b/cds_migrator_kit/rdm/records/transform/transform.py index bcb60b1b..07bc595e 100644 --- a/cds_migrator_kit/rdm/records/transform/transform.py +++ b/cds_migrator_kit/rdm/records/transform/transform.py @@ -134,7 +134,9 @@ def _transform_xml_to_json(self, raw_dump_entry): ) timestamp, dojson_entry = dump.latest_revision self.dojson_entry = dojson_entry - self.record_state_logger.add_record(dojson_entry) + # mappers pop the keys they consume off dojson_entry, which would + # otherwise strip the logged dump record. + self.record_state_logger.add_record(deepcopy(dojson_entry)) return dump def _parent(self, raw_dump_entry, record): diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py index 695a4e3a..114354b5 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/base.py @@ -931,6 +931,11 @@ def related_identifiers_787(self, key, value): "relation_type": {"id": "references"}, "resource_type": {"id": "publication-preprint"}, }, + "talk": { + # Used for when an article has a related video of a talk where that article is explained/demonstrated + "relation_type": {"id": "isdocumentedby"}, + "resource_type": {"id": "video"}, + }, } if recid: diff --git a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py index e1ac8b33..888c5292 100644 --- a/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py +++ b/cds_migrator_kit/rdm/records/transform/xml_processing/rules/research.py @@ -22,6 +22,9 @@ strip_output, ) from cds_migrator_kit.transform.xml_processing.quality.parsers import StringValue +from cds_migrator_kit.transform.xml_processing.rules.base import ( + extract_contributor_names, +) from ...config import ( udc_pattern, @@ -276,6 +279,28 @@ def journal(self, key, value): related_ids.append(isbn_related_id) self["related_identifiers"] = related_ids + if "m" in value: + # As discussed with SIS, we only ignore 773__m if there is only one in the record + # and its value is `publication` + if self.get("_773_m_seen"): + raise UnexpectedValue( + "Multiple 773__m seen. Record requires manual curation.", + subfield="m", + field=key, + value=value, + ) + + m_value = value.get("m") + if m_value != "publication": + raise UnexpectedValue( + f'Only value "publication" can be ignored for 773__m. Value "{m_value}" requires manual curation.', + subfield="m", + field=key, + value=value, + ) + + self["_773_m_seen"] = True + # p/n/v are journal-specific; c alone with w is a conference proceedings artid is_journal = any(f in value for f in ["p", "n", "v"]) is_journal_year = any(f in value for f in ["p", "n", "v", "c"]) @@ -290,6 +315,8 @@ def journal(self, key, value): if conference_url: identifiers.append({"scheme": "URL", "identifier": conference_url}) if conference_cnum: + # Some old records have slashes instead of hyphens in the INSPIRE conference cnum + conference_cnum = conference_cnum.replace("/", "-") identifiers.append({"scheme": "inspire", "identifier": conference_cnum}) new_meeting["identifiers"] = identifiers if conference_acronym: @@ -819,3 +846,39 @@ def ep_approval(self, key, value): }.items() if v } + + +@model.over("contributors", "^270__") +@for_each_value +def contact_person(self, key, value): + """Extract the contact persond details, mapping the name if it's available and the email otherwise.""" + contact_email = value.get("m") + contact_name = value.get("p") + + if contact_name is not None: + # The contact name takes precedence over the email + names = extract_contributor_names(contact_name) + return { + "person_or_org": {"type": "personal", **names}, + "role": {"id": "contactperson"}, + } + + if contact_email is not None: + if "@" not in contact_email: + raise UnexpectedValue( + "Value did not look like an email address", + subfield="m", + field=key, + value=value, + ) + + return { + "person_or_org": { + "type": "personal", + "name": contact_email, + "family_name": contact_email, + }, + "role": {"id": "contactperson"}, + } + + raise IgnoreKey("contributors") diff --git a/cds_migrator_kit/rdm/streams.yaml b/cds_migrator_kit/rdm/streams.yaml index c57e48f5..72192d37 100644 --- a/cds_migrator_kit/rdm/streams.yaml +++ b/cds_migrator_kit/rdm/streams.yaml @@ -634,6 +634,17 @@ records: missing_users: cds_migrator_kit/rdm/data/users communities_ids: - "88a105fe-4713-493b-b555-6ab398599d21" + na58: + data_dir: cds_migrator_kit/rdm/data/north_area/na58 + plots: true + create_inclusion_request: true + extract: + dirpath: cds_migrator_kit/rdm/data/north_area/na58/dump/ + transform: + files_dump_dir: cds_migrator_kit/rdm/data/north_area/na58/files/ + missing_users: cds_migrator_kit/rdm/data/users + communities_ids: + - "742fea6c-0685-4da2-8000-c1f19635ba86" na61: data_dir: cds_migrator_kit/rdm/data/north_area/na61 plots: true @@ -644,7 +655,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na61/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "cf52a005-cc57-4a69-a135-771bf9f28d8a" na62: data_dir: cds_migrator_kit/rdm/data/north_area/na62 plots: true @@ -655,7 +666,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na62/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "71045612-b90c-4dba-83a5-c5e55d5b0622" na63: data_dir: cds_migrator_kit/rdm/data/north_area/na63 plots: true @@ -666,7 +677,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na63/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "b018f5fc-66b3-47e9-af06-cb10f0021587" na64: data_dir: cds_migrator_kit/rdm/data/north_area/na64 plots: true @@ -677,7 +688,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na64/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "25a0fa11-6e27-419a-a00e-3b4b30ef0182" na65: data_dir: cds_migrator_kit/rdm/data/north_area/na65 plots: true @@ -688,7 +699,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na65/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "f6011761-fd7d-40dd-bba6-fcf6fa93edbc" na66: data_dir: cds_migrator_kit/rdm/data/north_area/na66 plots: true @@ -699,7 +710,7 @@ records: files_dump_dir: cds_migrator_kit/rdm/data/north_area/na66/files/ missing_users: cds_migrator_kit/rdm/data/users communities_ids: - - "" + - "2ab3ab07-a634-4de8-8489-f99bc7fabec2" comments: faser-drafts: dir_path: /migration/faser-drafts/comments/ diff --git a/cds_migrator_kit/transform/overdo.py b/cds_migrator_kit/transform/overdo.py index 38cdb070..6aef1bce 100644 --- a/cds_migrator_kit/transform/overdo.py +++ b/cds_migrator_kit/transform/overdo.py @@ -6,6 +6,7 @@ # the terms of the MIT License; see LICENSE file for more details. """CDS-RDM overdo model.""" + from copy import deepcopy from cds_dojson.overdo import Overdo @@ -80,6 +81,10 @@ def clean_missing(exc, output, key, value, rectype=None): items = iteritems(blob) items = sorted(items, key=lambda item: item[0]) for key, value in items: + # Skip completely empty datafields that have no subfields + # Some records might have these for unknown reasons (e.g. https://cds.cern.ch/record/2964767/export/xm?ln=en) + if isinstance(value, dict) and not value.keys(): + continue try: result = self.index.query(key) if not result: diff --git a/cds_migrator_kit/transform/xml_processing/rules/base.py b/cds_migrator_kit/transform/xml_processing/rules/base.py index 0bee20ef..d6b8d176 100644 --- a/cds_migrator_kit/transform/xml_processing/rules/base.py +++ b/cds_migrator_kit/transform/xml_processing/rules/base.py @@ -7,7 +7,6 @@ """CDS-RDM migration rules module.""" - import pycountry from cds_dojson.marc21.fields.utils import out_strip from dojson.errors import IgnoreKey @@ -100,6 +99,18 @@ def languages(self, key, value): raise UnexpectedValue(field=key, subfield="a") +def extract_contributor_names(names): + """Convert a single contributor string to family and potentially given name.""" + names = names.strip().split(",") + + if len(names) == 2: + names = {"family_name": names[0].strip(), "given_name": names[1].strip()} + else: + names = {"family_name": names[0].strip()} + + return names + + def process_contributors(key, value, orcid_subfield="k"): """Utility processing contributors XML.""" role = value.get("e") @@ -117,7 +128,9 @@ def process_contributors(key, value, orcid_subfield="k"): _affiliations = force_list(value.get("t", "")) affiliations = [] # just to avoid the missing rule exception - text = value.get("u") or value.get("v") + text_u = value.get("u") + text_v = value.get("v") + text = text_u or text_v grid_value = None for aff in _affiliations: if aff: @@ -148,12 +161,8 @@ def process_contributors(key, value, orcid_subfield="k"): if type(names) == tuple or names is None: raise UnexpectedValue(field=key, subfield="a", value=names) - names = names.strip().split(",") + names = extract_contributor_names(names) - if len(names) == 2: - names = {"family_name": names[0].strip(), "given_name": names[1].strip()} - else: - names = {"family_name": names[0].strip()} contributor = { "person_or_org": { "type": "personal", diff --git a/tests/cds-rdm/conftest.py b/tests/cds-rdm/conftest.py index 8cf329e7..059513f5 100644 --- a/tests/cds-rdm/conftest.py +++ b/tests/cds-rdm/conftest.py @@ -877,9 +877,9 @@ def experiments_v(app, exp_type): vocab = vocabulary_service.create( system_identity, { - "id": "COMPASS NA58", + "id": "NA58", "title": { - "en": "COMPASS NA58", + "en": "NA58", }, "props": {"link": "http://bla.web.cern.ch/lhcb/"}, "type": "experiments", @@ -1448,6 +1448,15 @@ def contributors_role_v(app, contributors_role_type): "type": "contributorsroles", }, ) + vocabulary_service.create( + system_identity, + { + "id": "contactperson", + "props": {"datacite": "ContactPerson"}, + "title": {"en": "Contact person"}, + "type": "contributorsroles", + }, + ) Vocabulary.index.refresh() diff --git a/tests/cds-rdm/data/vocabularies/experiments.yaml b/tests/cds-rdm/data/vocabularies/experiments.yaml index 9843e598..4c0c4fdf 100644 --- a/tests/cds-rdm/data/vocabularies/experiments.yaml +++ b/tests/cds-rdm/data/vocabularies/experiments.yaml @@ -15,9 +15,9 @@ - id: RP title: en: RP -- id: COMPASS NA58 +- id: NA58 title: - en: COMPASS NA58 + en: NA58 - id: I216 title: en: I216 diff --git a/tests/cds-rdm/test_base_rules.py b/tests/cds-rdm/test_base_rules.py index 19759a06..70c3034c 100644 --- a/tests/cds-rdm/test_base_rules.py +++ b/tests/cds-rdm/test_base_rules.py @@ -441,3 +441,25 @@ def test_imprint_info_269_invalid_date_raises_error(self): record = {"custom_fields": {}} with pytest.raises(UnexpectedValue): imprint_info(record, "269__", {"c": "not-a-valid-date"}) + + +class TestEmptyDatafields: + """Test that datafields without any subfields are skipped.""" + + def test_empty_datafield_is_ignored(self): + """Test that an empty 540 is skipped while a sibling 540 is kept.""" + from cds_dojson.marc21.utils import create_record + + from cds_migrator_kit.rdm.records.transform.models.base_record import ( + rdm_base_record_model, + ) + + marc_record = create_record( + '12345' + '' + '' + 'CC-BY' + "" + ) + result = rdm_base_record_model.do(marc_record) + assert result["rights"] == [{"title": {"en": "CC-BY"}}] diff --git a/tests/cds-rdm/test_full_migration.py b/tests/cds-rdm/test_full_migration.py index ee45fa23..eaaa8161 100644 --- a/tests/cds-rdm/test_full_migration.py +++ b/tests/cds-rdm/test_full_migration.py @@ -371,9 +371,9 @@ def irregular_exp_field(record): ], "cern:experiments": [ { - "id": "COMPASS NA58", + "id": "NA58", "title": { - "en": "COMPASS NA58", + "en": "NA58", }, }, ],