Skip to content
4 changes: 3 additions & 1 deletion cds_migrator_kit/rdm/records/transform/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@
"aip",
"jacow",
]
KEYWORD_SCHEMES_TO_DROP = ["proquest", "disxa", "inspeq"]
KEYWORD_SCHEMES_TO_DROP = ["proquest", "disxa", "inspeq", "jinr"]

ALLOWED_THESIS_COLLECTIONS = [
"thesis",
Expand Down Expand Up @@ -162,4 +162,6 @@
# Legacy experiment names remapped to vocabulary ids before lookup
EXPERIMENT_ALIASES = {
"t2k": "re13",
"compass": "NA58",
"compass na58": "NA58",
}
Original file line number Diff line number Diff line change
Expand Up @@ -232,6 +232,9 @@ class JournalMapper(CustomFieldMapper):

def apply(self, ctx):
"""Set ctx.custom_fields["journal:journal"]."""
# `_773_m_seen` is an internal bookkeeping field, just to check that we only have one 773__m
# per record as agreed with SIS.
ctx.dojson_entry.pop("_773_m_seen", None)
journal = ctx.dojson_entry.get("custom_fields", {}).get("journal:journal", {})
if journal and not journal.get("title"):
ctx.flag_curation(
Expand Down
32 changes: 18 additions & 14 deletions cds_migrator_kit/rdm/records/transform/models/_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,34 +7,36 @@
"0248_q",
"852__c", # holdings will be taken separately
"852__h",
# "035__h", # OAI harvest tag or timestamp
# "035__d", # OAI harvest tag or timestamp
# "035__m", # OAI harvest format (e.g. `marcxml`)
# "035__t", # oai harvest tag
# "035__u", # oai harvest tag
# "035__z", # oai harvest tag
"035__h", # OAI harvest tag or timestamp
"035__d", # OAI harvest tag or timestamp
"035__m", # OAI harvest format (e.g. `marcxml`)
"035__t", # oai harvest tag
"035__u", # oai harvest tag
"035__z", # oai harvest tag
"037__c", # arxiv subject
"100__m", # email of contributor
"245__9", # Provenance of title
# "270__m", # Contact email
"300__a", # number of pages
"340__a", # See decision log
"500__9", # Provenance of the note
"520__9", # Provenance of the description
# "540__3", # Material of the license
# "540__9", # Also material of the license
# "542__3", # Also material of the license
"540__3", # Material of the license
"540__9", # Also material of the license
"540__g", # From SIS (see decision log)
"542__3", # Also material of the license
"700__m", # email of contributor
# "773__t", # from SIS: can be ignored
# "773__0", # from SIS: can be ignored
# "773__o", # from SIS: can be ignored
# "773__x", # INSPIRE publication note
"773__t", # from SIS: can be ignored
"773__0", # from SIS: can be ignored
"773__o", # from SIS: can be ignored
"773__x", # INSPIRE publication note
"8564_8", # file id
"8564_s", # bibdoc id
"8564_x", # icon thumbnails sizes
"8564_y", # file description - done by files dump
"8564_8", # File information (done by file dump)
"8564_q", # File links File information (done by file dump)
"8564_z", # Websubmit "stamp" (migrated as file metadata)
"905__m", # Submitter email address
"916__y", # year, redundant value
"937__c", # last modified by
"937__s", # last modification date
Expand Down Expand Up @@ -68,4 +70,6 @@
"999C6t", # https://cds.cern.ch/record/2284606/export/hm?ln=en
"999C6v", # https://cds.cern.ch/record/2284606/export/hm?ln=en
"999C5d", # old INSPIRE attr
"999C69", # See decision log
"999C6c", # See decision log
}
15 changes: 10 additions & 5 deletions cds_migrator_kit/rdm/records/transform/models/north_area.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
# CDS-RDM is free software; you can redistribute it and/or modify it under
# the terms of the MIT License; see LICENSE file for more details.

"""CDS-RDM North Area models (NA61-64)."""
"""CDS-RDM North Area models (NA58 & NA61-66)."""

from cds_migrator_kit.rdm.records.transform.models._config import IGNORE_SYSTEM_KEYS
from cds_migrator_kit.rdm.records.transform.models.base_publication_record import (
Expand All @@ -17,14 +17,19 @@
class NorthAreaModel(CdsOverdo):
"""Translation model for North Area experiments."""

__query__ = """693__.e:"NA61" OR 693__.e:"SHINE NA61" OR 693__.e:"NA62" OR 693__.e:"NA63" OR 693__.e:"NA64" OR 693__.e:"DsTau NA65" OR 693__.e:"AMBER NA66"
-980__:THESIS -980__:DELETED -980__:HIDDEN -980__:DUMMY"""
__query__ = """
693__.e:"COMPASS" OR 693__.e:"COMPASS NA58" OR 693__.e:"NA61" OR 693__.e:"SHINE NA61" OR
Comment thread
palkerecsenyi marked this conversation as resolved.
693__.e:"NA62" OR 693__.e:"NA63" OR 693__.e:"NA64" OR 693__.e:"DsTau NA65" OR 693__.e:"AMBER NA66"
-037__:CERN-STUDENTS-Note-* -690C_:SCICOM -980__:THESIS -980__:DELETED -980__:HIDDEN -980__:DUMMY
"""

__ignore_keys__ = IGNORE_SYSTEM_KEYS | {
"270__m", # Email of contact person
"500__9", # Provenance of the note
"595_Da", # From SIS: these can be ignored
"595_Dd", # From SIS: these can be ignored
"595_Ds", # From SIS: these can be ignored
"595__9", # From SIS: these can be ignored
"903__s", # 'public'
"905__m", # Submitter email address
"995__a", # "Inspire"
}

Expand Down
1 change: 0 additions & 1 deletion cds_migrator_kit/rdm/records/transform/models/research.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,6 @@ class ResearchModel(CdsOverdo):
"037__c", # arxiv subject
"100__m", # email of contributor
"245__9", # title provenance
"270__m", # document contact email
"300__a", # number of pages
"340__a", # TODO ignore material?
"540__3", # TODO still ignore the material of the license?
Expand Down
4 changes: 3 additions & 1 deletion cds_migrator_kit/rdm/records/transform/transform.py
Original file line number Diff line number Diff line change
Expand Up @@ -134,7 +134,9 @@ def _transform_xml_to_json(self, raw_dump_entry):
)
timestamp, dojson_entry = dump.latest_revision
self.dojson_entry = dojson_entry
self.record_state_logger.add_record(dojson_entry)
# mappers pop the keys they consume off dojson_entry, which would
# otherwise strip the logged dump record.
self.record_state_logger.add_record(deepcopy(dojson_entry))
return dump

def _parent(self, raw_dump_entry, record):
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -931,6 +931,11 @@ def related_identifiers_787(self, key, value):
"relation_type": {"id": "references"},
"resource_type": {"id": "publication-preprint"},
},
"talk": {
# Used for when an article has a related video of a talk where that article is explained/demonstrated
"relation_type": {"id": "isdocumentedby"},
"resource_type": {"id": "video"},
Comment thread
palkerecsenyi marked this conversation as resolved.
},
}

if recid:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,9 @@
strip_output,
)
from cds_migrator_kit.transform.xml_processing.quality.parsers import StringValue
from cds_migrator_kit.transform.xml_processing.rules.base import (
extract_contributor_names,
)

from ...config import (
udc_pattern,
Expand Down Expand Up @@ -276,6 +279,28 @@ def journal(self, key, value):
related_ids.append(isbn_related_id)
self["related_identifiers"] = related_ids

if "m" in value:
# As discussed with SIS, we only ignore 773__m if there is only one in the record
# and its value is `publication`
if self.get("_773_m_seen"):
raise UnexpectedValue(
"Multiple 773__m seen. Record requires manual curation.",
subfield="m",
field=key,
value=value,
)

m_value = value.get("m")
if m_value != "publication":
raise UnexpectedValue(
f'Only value "publication" can be ignored for 773__m. Value "{m_value}" requires manual curation.',
subfield="m",
field=key,
value=value,
)

self["_773_m_seen"] = True

# p/n/v are journal-specific; c alone with w is a conference proceedings artid
is_journal = any(f in value for f in ["p", "n", "v"])
is_journal_year = any(f in value for f in ["p", "n", "v", "c"])
Expand All @@ -290,6 +315,8 @@ def journal(self, key, value):
if conference_url:
identifiers.append({"scheme": "URL", "identifier": conference_url})
if conference_cnum:
# Some old records have slashes instead of hyphens in the INSPIRE conference cnum
conference_cnum = conference_cnum.replace("/", "-")
identifiers.append({"scheme": "inspire", "identifier": conference_cnum})
new_meeting["identifiers"] = identifiers
if conference_acronym:
Expand Down Expand Up @@ -819,3 +846,39 @@ def ep_approval(self, key, value):
}.items()
if v
}


@model.over("contributors", "^270__")
@for_each_value
def contact_person(self, key, value):
"""Extract the contact persond details, mapping the name if it's available and the email otherwise."""
contact_email = value.get("m")
contact_name = value.get("p")

if contact_name is not None:
# The contact name takes precedence over the email
names = extract_contributor_names(contact_name)
return {
"person_or_org": {"type": "personal", **names},
"role": {"id": "contactperson"},
}

if contact_email is not None:
if "@" not in contact_email:
raise UnexpectedValue(
"Value did not look like an email address",
subfield="m",
field=key,
value=value,
)

return {
"person_or_org": {
"type": "personal",
"name": contact_email,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Can we try not to store emails in this field since we should avoid exposing them for data privacy reasons?
Could you check with Fleur on what they think?
I would try to deconstruct the email to extract name.surname pattern...

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

did we check if we have these emails in the system? maybe they are existing users

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

for the authors we don't need to do this, but yes, we could double check to have more accurate data

"family_name": contact_email,
},
"role": {"id": "contactperson"},
}

raise IgnoreKey("contributors")
23 changes: 17 additions & 6 deletions cds_migrator_kit/rdm/streams.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -634,6 +634,17 @@ records:
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- "88a105fe-4713-493b-b555-6ab398599d21"
na58:
data_dir: cds_migrator_kit/rdm/data/north_area/na58
plots: true
create_inclusion_request: true
extract:
dirpath: cds_migrator_kit/rdm/data/north_area/na58/dump/
transform:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na58/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- "742fea6c-0685-4da2-8000-c1f19635ba86"
na61:
data_dir: cds_migrator_kit/rdm/data/north_area/na61
plots: true
Expand All @@ -644,7 +655,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na61/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "cf52a005-cc57-4a69-a135-771bf9f28d8a"
na62:
data_dir: cds_migrator_kit/rdm/data/north_area/na62
plots: true
Expand All @@ -655,7 +666,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na62/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "71045612-b90c-4dba-83a5-c5e55d5b0622"
na63:
data_dir: cds_migrator_kit/rdm/data/north_area/na63
plots: true
Expand All @@ -666,7 +677,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na63/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "b018f5fc-66b3-47e9-af06-cb10f0021587"
na64:
data_dir: cds_migrator_kit/rdm/data/north_area/na64
plots: true
Expand All @@ -677,7 +688,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na64/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "25a0fa11-6e27-419a-a00e-3b4b30ef0182"
na65:
data_dir: cds_migrator_kit/rdm/data/north_area/na65
plots: true
Expand All @@ -688,7 +699,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na65/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "f6011761-fd7d-40dd-bba6-fcf6fa93edbc"
na66:
data_dir: cds_migrator_kit/rdm/data/north_area/na66
plots: true
Expand All @@ -699,7 +710,7 @@ records:
files_dump_dir: cds_migrator_kit/rdm/data/north_area/na66/files/
missing_users: cds_migrator_kit/rdm/data/users
communities_ids:
- ""
- "2ab3ab07-a634-4de8-8489-f99bc7fabec2"
comments:
faser-drafts:
dir_path: /migration/faser-drafts/comments/
Expand Down
5 changes: 5 additions & 0 deletions cds_migrator_kit/transform/overdo.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
# the terms of the MIT License; see LICENSE file for more details.

"""CDS-RDM overdo model."""

from copy import deepcopy

from cds_dojson.overdo import Overdo
Expand Down Expand Up @@ -80,6 +81,10 @@ def clean_missing(exc, output, key, value, rectype=None):
items = iteritems(blob)
items = sorted(items, key=lambda item: item[0])
for key, value in items:
# Skip completely empty datafields that have no subfields
# Some records might have these for unknown reasons (e.g. https://cds.cern.ch/record/2964767/export/xm?ln=en)
if isinstance(value, dict) and not value.keys():
continue
try:
result = self.index.query(key)
if not result:
Expand Down
23 changes: 16 additions & 7 deletions cds_migrator_kit/transform/xml_processing/rules/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,6 @@

"""CDS-RDM migration rules module."""


import pycountry
from cds_dojson.marc21.fields.utils import out_strip
from dojson.errors import IgnoreKey
Expand Down Expand Up @@ -100,6 +99,18 @@ def languages(self, key, value):
raise UnexpectedValue(field=key, subfield="a")


def extract_contributor_names(names):
"""Convert a single contributor string to family and potentially given name."""
names = names.strip().split(",")

if len(names) == 2:
names = {"family_name": names[0].strip(), "given_name": names[1].strip()}
else:
names = {"family_name": names[0].strip()}

return names


def process_contributors(key, value, orcid_subfield="k"):
"""Utility processing contributors XML."""
role = value.get("e")
Expand All @@ -117,7 +128,9 @@ def process_contributors(key, value, orcid_subfield="k"):
_affiliations = force_list(value.get("t", ""))
affiliations = []
# just to avoid the missing rule exception
text = value.get("u") or value.get("v")
text_u = value.get("u")
text_v = value.get("v")
text = text_u or text_v
grid_value = None
for aff in _affiliations:
if aff:
Expand Down Expand Up @@ -148,12 +161,8 @@ def process_contributors(key, value, orcid_subfield="k"):
if type(names) == tuple or names is None:
raise UnexpectedValue(field=key, subfield="a", value=names)

names = names.strip().split(",")
names = extract_contributor_names(names)

if len(names) == 2:
names = {"family_name": names[0].strip(), "given_name": names[1].strip()}
else:
names = {"family_name": names[0].strip()}
contributor = {
"person_or_org": {
"type": "personal",
Expand Down
Loading
Loading