Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
862 changes: 862 additions & 0 deletions cds_migration_progress.html

Large diffs are not rendered by default.

15 changes: 14 additions & 1 deletion cds_migrator_kit/rdm/migration_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -535,7 +535,10 @@ def resolve_record_pid(pid):

### EP Approval configuration only needed for local, it should use cds-rdm config for de/sandbox/prod
# ===========================
CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "78b3c4aa-c4e6-4502-8226-67ba2d347afe"
# ATTENTION: please don't modify this local value - the community is created
# via cds-rdm fixtures with this id - if you have another ID locally
# change it in your local db
CDS_CERN_SCIENTIFIC_COMMUNITY_ID = "c2c46ab3-5fb4-4d86-83c6-5d9dc8392d6f"
"""The id of the CERN Scientific community.

This is only a local-dev default: on other instances (sandbox/prod), set the
Expand Down Expand Up @@ -626,4 +629,14 @@ def resolve_record_pid(pid):
"counter_digits": 3, # zero-padding width, e.g. 3 → "001"
},
},
"6a289642-5378-4daf-87b5-bb58af00487a": {
# DIRAC
"label": "EP approval", # shown in UI buttons/headings
"referee_group": "cds-ph-ep-publications-referee-non-lhc", # CERN e-group slug
"report_number": {
"prefix": "CERN-EP", # literal prefix, e.g. "CERN-EP"
"include_year": True, # append the current year after prefix
"counter_digits": 3, # zero-padding width, e.g. 3 → "001"
},
},
}
18 changes: 17 additions & 1 deletion cds_migrator_kit/rdm/records/load/load.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
from invenio_db import db
from invenio_db.uow import ModelCommitOp, UnitOfWork
from invenio_i18n import _
from invenio_pidstore.errors import PIDDoesNotExistError
from invenio_pidstore.errors import PIDAlreadyExists, PIDDoesNotExistError
from invenio_pidstore.models import PersistentIdentifier
from invenio_rdm_migrator.load.base import Load
from invenio_rdm_records.proxies import current_rdm_records_service
Expand Down Expand Up @@ -308,6 +308,22 @@ def _load(self, entry: MigrationEntry):
# apply after record fully finished (does not sync at the spot, only enabled)
self._apply_clc_sync(recid_state_after_load, entry)
return recid_state_after_load
except PIDAlreadyExists:
# The legacy recid's `lrecid` PID was minted by someone else
# between our _should_skip_recid() check above and now - e.g. a
# concurrently running migration of another collection whose
# dump cross-lists the same legacy recid (two former-experiment
# collections can both ship the same record). Treat it exactly
# like _should_skip_recid: already migrated, nothing to do.
# this happens when you run several mirations at the same time
self.migration_logger.add_information(
recid,
state={
"message": "Record already migrated (lrecid PID already minted)",
"value": recid,
},
)
self.migration_logger.finalise_record(recid)
except (UnexpectedValue, ManualImportRequired, GrantCreationError) as e:
self.migration_logger.add_log(e, record=entry)
except (CDSMigrationException, ValidationError, InvalidRelationValue) as e:
Expand Down
12 changes: 11 additions & 1 deletion cds_migrator_kit/rdm/records/transform/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -161,5 +161,15 @@

# Legacy experiment names remapped to vocabulary ids before lookup
EXPERIMENT_ALIASES = {
"t2k": "re13",
"t2k": "RE13",
"antares": "RE6",
"dirac": "PS212",
"dirac ps212": "PS212",
"harp ps214": "PS214",
"harp": "PS214",
"dampe": "RE29",
}

# 693__e values that are keywords rather than real experiments (curated as
# "mots clef"), mapped to subjects instead of cern:experiments
EXPERIMENTS_AS_SUBJECTS = ["d3", "r104", "r105a"]
8 changes: 5 additions & 3 deletions cds_migrator_kit/rdm/records/transform/entities/record.py
Original file line number Diff line number Diff line change
Expand Up @@ -168,7 +168,7 @@ def _files(self, dump):
files = dump.files
return {"enabled": bool(files)}

def _metadata(self, dojson_entry, raw_dump_entry):
def _metadata(self, dojson_entry, raw_dump_entry, pids=None):
"""Build the metadata dict by running the composed field mappers.

Whether every ``dojson_entry`` key ended up consumed *somewhere* in
Expand All @@ -183,6 +183,7 @@ def _metadata(self, dojson_entry, raw_dump_entry):
raw_dump_entry=raw_dump_entry,
migration_logger=self.migration_logger,
affiliations_mapping=self.affiliations_mapping,
pids=pids or {},
)
metadata = ctx.metadata
# Order matters: ResourceTypeMapper must run before TitleMapper reads
Expand Down Expand Up @@ -259,10 +260,11 @@ def build(self) -> RecordBody:
# same reason as _pids()/_access()'s record_restriction pop.
internal_notes = dojson_entry.pop("internal_notes", None)

pids = self._pids(dojson_entry)
record_json_output = {
"files": self._files(dump),
"pids": self._pids(dojson_entry),
"metadata": self._metadata(dojson_entry, raw_dump_entry),
"pids": pids,
"metadata": self._metadata(dojson_entry, raw_dump_entry, pids),
"internal_notes": internal_notes,
"custom_fields": custom_fields,
}
Expand Down
1 change: 1 addition & 0 deletions cds_migrator_kit/rdm/records/transform/mappers/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@ class RecordTransformContext:
migration_logger: object = None
affiliations_mapping: object = None
access_grants_view: object = None
pids: dict = field(default_factory=dict)
metadata: dict = field(default_factory=dict)
custom_fields: dict = field(default_factory=dict)

Expand Down
18 changes: 16 additions & 2 deletions cds_migrator_kit/rdm/records/transform/mappers/custom_fields.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,10 @@
RecordFlaggedCuration,
UnexpectedValue,
)
from cds_migrator_kit.rdm.records.transform.config import EXPERIMENT_ALIASES
from cds_migrator_kit.rdm.records.transform.config import (
EXPERIMENT_ALIASES,
EXPERIMENTS_AS_SUBJECTS,
)
from cds_migrator_kit.rdm.records.transform.mappers.base import CustomFieldMapper
from cds_migrator_kit.rdm.records.transform.mappers.vocabulary import search_vocabulary

Expand All @@ -34,6 +37,12 @@ def apply(self, ctx):
for experiment in experiments:
if experiment.lower().strip() in ["not applicable", "xx"]:
continue
if experiment.lower().strip() in EXPERIMENTS_AS_SUBJECTS:
# curated as keywords ("mots clef"), not real experiments
ctx.dojson_entry.setdefault("subjects", []).append(
{"subject": experiment}
)
continue
experiment = EXPERIMENT_ALIASES.get(experiment.lower().strip(), experiment)
result = search_vocabulary(experiment, "experiments")
if result and result not in experiments_out:
Expand Down Expand Up @@ -153,7 +162,12 @@ def apply(self, ctx):
"cern:accelerators", []
)
for accelerator in accelerators:
if accelerator.lower().strip() in ["not applicable", "xx", "fermi"]:
if accelerator.lower().strip() in [
"not applicable",
"xx",
"fermi",
"cern recognized expt.",
]:
continue
result = search_vocabulary(accelerator, "accelerators")
if result and result not in accelerators_out:
Expand Down
55 changes: 52 additions & 3 deletions cds_migrator_kit/rdm/records/transform/mappers/metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -189,12 +189,23 @@ def map_value(self, ctx):


class TableOfContentsMapper(FieldMapper):
"""Folds table_of_content into additional_descriptions."""
"""Folds table_of_content into additional_descriptions.

Also the single place where the final ``additional_descriptions`` list
is deduplicated: many different dojson rules append to it (520/246/
035/500/210/... across base.py and the various collection-specific
rule modules), some legacy records repeat the very same MARC field
(identical text, sometimes only differing in a provenance subfield
nothing here reads), and not every one of those rules remembers to
guard against re-adding an entry already present. Deduplicating once
here, after every rule has run, doesn't depend on each of them getting
that guard right.
"""

id = "additional_descriptions"

def map_value(self, ctx):
"""Move table_of_content into additional_descriptions and return it."""
"""Move table_of_content into additional_descriptions and dedupe."""
dojson_entry = ctx.dojson_entry
toc = dojson_entry.get("table_of_content", [])
additional_desc = dojson_entry.get("additional_descriptions", [])
Expand All @@ -204,6 +215,14 @@ def map_value(self, ctx):
)
dojson_entry["additional_descriptions"] = additional_desc
dojson_entry.pop("table_of_content")

deduped = []
for description in dojson_entry.get("additional_descriptions", []):
if description not in deduped:
deduped.append(description)
if deduped:
dojson_entry["additional_descriptions"] = deduped

return dojson_entry.get("additional_descriptions")


Expand Down Expand Up @@ -244,6 +263,37 @@ def map_value(self, ctx):
return identifiers


#: `setlink` is a CDS-internal redirector, not a real related resource -
#: drop any related_identifiers entry pointing at it.
_SETLINK_URL_PREFIX = "http://documents.cern.ch/cgi-bin/setlink?"


class RelatedIdentifiersMapper(FieldMapper):
"""Maps related_identifiers, dropping CDS-internal setlink URLs."""

id = "related_identifiers"

def map_value(self, ctx):
"""Return related_identifiers without setlink URLs or the record's own DOI."""
related_identifiers = ctx.dojson_entry.get("related_identifiers", [])
record_doi = ((ctx.pids or {}).get("doi") or {}).get("identifier")
record_doi = record_doi.strip().lower() if record_doi else None
return [
item
for item in related_identifiers
if not (
(item.get("scheme") or "").upper() == "URL"
and (item.get("identifier") or "").startswith(_SETLINK_URL_PREFIX)
)
# the record's own DOI is already in pids, don't repeat it
and not (
record_doi
and (item.get("scheme") or "").lower() == "doi"
and (item.get("identifier") or "").strip().lower() == record_doi
)
]


# Fields that pass through unchanged from dojson_entry - kept explicit in the
# composed list (mappers/config equivalent) rather than open-ended, so the
# "forgotten metadata key" completeness check in
Expand All @@ -256,7 +306,6 @@ def map_value(self, ctx):
"languages",
"dates",
"funding",
"related_identifiers",
"rights",
"copyright",
)
2 changes: 2 additions & 0 deletions cds_migrator_kit/rdm/records/transform/mappers/registry.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@
PASSTHROUGH_METADATA_FIELDS,
IdentifiersMapper,
PublicationDateMapper,
RelatedIdentifiersMapper,
ResourceTypeMapper,
SubjectsMapper,
TableOfContentsMapper,
Expand All @@ -45,6 +46,7 @@
PublicationDateMapper(),
SubjectsMapper(),
IdentifiersMapper(),
RelatedIdentifiersMapper(),
*(PassthroughMapper(field_name) for field_name in PASSTHROUGH_METADATA_FIELDS),
)

Expand Down
1 change: 1 addition & 0 deletions cds_migrator_kit/rdm/records/transform/models/lep.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ class LEPResearchModel(ResearchModel):

__ignore_keys__ = {
"594__a", # can be ignored for this collection
"852__a", # location
"775__p", # can be ignored for this collection - title of another volume
"775__c", # year of volume
"596__a", # multivolume tag
Expand Down
3 changes: 2 additions & 1 deletion cds_migrator_kit/rdm/records/transform/models/research.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@
class ResearchModel(CdsOverdo):
"""Translation model for research."""

__query__ = '693__.e:"DAMPE RE29" OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM'
__query__ = '693__.e:"DAMPE RE29" OR 693__.e:RE29 OR 693__.e:DAMPE OR 037__:DIRAC-NOTE* OR 037__:DIRAC-Note* OR 037__:DIRAC-CONF* OR 037__:DIRAC-DOC* OR 037__:DIRAC-PUB* OR 693__:UA2 OR 693__:UA4 OR 693__:UA5 OR 693__:UA8 OR 980__:INTNOTEHARPCDPPUBL OR 980__:PRIVIMXGAM -980__:THESIS -037__:CERN-STUDENTS-Note-* -980__:DELETED -980__.a:DUMMY -690C_.a:SCICOM -980:BULLETINNEWS'

__ignore_keys__ = {
"0248_a",
Expand Down Expand Up @@ -50,6 +50,7 @@ class ResearchModel(CdsOverdo):
"542__8", # agreed not to migrate open access related fields
"595__i", # TODO ??
"695__e", # some inspire tag
"695__9", # bibclassify
"700__m", # email of contributor
"700__q", # TODO ignore? aliteration of the name, used for searching
"700__v", # TODO drop?
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -42,12 +42,15 @@ class ResearchCommitteeModel(CdsOverdo):
"340__a", # TODO ignore material?
"540__3", # TODO still ignore the material of the license?
"542__3", # TODO still ignore the material of the license?
"594__a", # ATN tag
"595__i", # TODO ??
"695__e", # some inspire tag
"695__9", # some inspire tag
"700__m", # email of contributor
"700__q", # TODO ignore? aliteration of the name, used for searching
"700__v", # TODO drop?
"773__x", # INSPIRE publication note
"852__a",
"8564_8", # file id
"8564_s", # bibdoc id
"8564_x", # icon thumbnails sizes
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -333,6 +333,9 @@ def report_number(self, key, value):
raise IgnoreKey("related_identifiers")
elif scheme.upper().startswith("B00"):
raise IgnoreKey("related_identifiers")
elif key == "088__" and scheme.upper().startswith("SC000"):
# internal scanning/digitisation request number, to drop
raise IgnoreKey("related_identifiers")
elif scheme.startswith("SCOO"):
identifier = scheme
scheme = "other"
Expand Down Expand Up @@ -828,6 +831,20 @@ def series_information(self, key, value):
return {"description": series, "type": {"id": "series-information"}}


@model.over("additional_descriptions", "^336__")
@for_each_value
def multiple_videos_note(self, key, value):
"""Translate the video-system's "multiple videos" cross-reference note.

The only recognised 336__a content - any other value is unexpected and
flagged for manual curation rather than silently dropped.
"""
note = StringValue(value.get("a", "")).parse()
if not note.startswith("Multiple videos have been identified with recid"):
raise UnexpectedValue(field=key, subfield="a", value=value)
return {"description": note, "type": {"id": "technical-info"}}


@model.over("related_identifiers", "^084__")
@for_each_value
def yellow_reports(self, key, value):
Expand Down Expand Up @@ -924,13 +941,17 @@ def related_identifiers_787(self, key, value):
"resource_type": {"id": "publication-report"},
},
"complemented by": {
"relation_type": {"id": "issuplementedby"},
"relation_type": {"id": "issupplementedby"},
"resource_type": {"id": "publication-report"},
},
"preprint": {
"relation_type": {"id": "references"},
"resource_type": {"id": "publication-preprint"},
},
"related video": {
"relation_type": {"id": "references"},
"resource_type": {"id": "video"},
},
}

if recid:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -748,6 +748,9 @@ def resource_type(self, key, value):
"lhcf_proc": {"id": "publication-conferenceproceeding"},
"lhcf_reports": {"id": "publication-report"},
"conferencepapers": {"id": "publication-conferencepaper"},
"technical note": {"id": "publication-technicalnote"},
"minutes": {"id": "publication-meetingminutes"},
"presentation": {"id": "presentation"},
}

try:
Expand Down
Loading
Loading