Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
86 changes: 6 additions & 80 deletions cds_migrator_kit/rdm/records/transform/models/isolde.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
from cds_migrator_kit.rdm.records.transform.models.base_publication_record import (
rdm_base_publication_model,
)
from cds_migrator_kit.rdm.records.transform.models.research import ResearchModel
from cds_migrator_kit.transform.overdo import CdsOverdo


Expand All @@ -18,86 +19,11 @@ class ISOLDEModel(CdsOverdo):

__query__ = '693__.a:"CERN ISOLDE" AND (980__:ARTICLE OR 980__:PREPRINT OR 980__:conferencepaper OR 980__:NOTE OR 980__:REPORT) -980__:DELETED -980__:DUMMY'

__ignore_keys__ = {
"0247_9", # provenance of the DOI
"0248_a",
"0248_p",
"0248_q",
"035__d", # oai harvest tag
"035__h", # oai harvest tag
"035__m", # oai harvest tag
"035__t", # oai harvest tag
"035__u", # oai harvest tag
"035__z", # oai harvest tag
"030__a", # CODEN journal code (e.g. "Phys. Lett., B") - obsolete identifier system, journal title already captured in 773__p
"037__c", # arxiv subject
"100__m", # email of contributor
"245__9", # title source
"270__m", # contact person email
"300__a", # number of pages
"336__a", # redundant field
"500__9", # provenance of the note
"520__9", # provenance of the description
"520__h", # provenance of the description
"540__3", # material of license
"540__9", # material of license
"542__3", # material of copyrights
"595__i",
"695__e", # inspire tag
"700__m", # email of contributor
"700__q", # aliteration of the name, used for searching
"700__v",
"773__0", # from SIS: can be ignored
"773__o", # from SIS: can be ignored
"773__t", # INSPIRE publication note
"773__x", # INSPIRE publication note
"773__z", # from SIS: can be ignored
"8564_8", # file id
"8564_s", # bibdoc id
"8564_w", # system field
"8564_x", # icon thumbnails sizes
"8564_y", # file description - done by files dump
"8564_z", # file comment, migrated via file metadata
"913__a", # citation
"913__c", # citation
"913__t", # citation
"913__v", # citation
"913__y", # citation
"916__y", # year, redundant value
"937__c", # last modified by
"937__s", # last modification date
"960__a", # base number
"961__c",
"961__h",
"961__l",
"961__x",
"964__a",
"980__b", # additional article tag
"981__a", # duplicate record id
"999C50",
"999C52",
"999C59",
"999C5a",
"999C5c",
"999C5h",
"999C5i",
"999C5k",
"999C5l",
"999C5m",
"999C5o",
"999C5p",
"999C5r",
"999C5s",
"999C5t",
"999C5u",
"999C5v",
"999C5x",
"999C5y",
"999C5z",
"999C6a",
"999C6t",
"999C6v",
}
# ResearchModel.__ignore_keys__ covers the shared publication baseline
# (035__ oai tags, 852__h/c holdings, 999C5*/6* citations, 8564_ file
# subfields, 773__ SIS fields, contributor emails, etc.).
# Only ISOLDE-specific additions are listed here.
__ignore_keys__ = ResearchModel.__ignore_keys__

_default_fields = {
"custom_fields": {},
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
# -*- coding: utf-8 -*-
#
# Copyright (C) 2026 CERN.
#
# CDS-RDM is free software; you can redistribute it and/or modify it under
# the terms of the MIT License; see LICENSE file for more details.

"""CDS-RDM ISOLDE migration rules."""

from dojson.errors import IgnoreKey

from cds_migrator_kit.errors import UnexpectedValue
from cds_migrator_kit.transform.xml_processing.quality.decorators import for_each_value
from cds_migrator_kit.transform.xml_processing.quality.parsers import StringValue

from ...models.isolde import isolde_model as model
from .base import identifiers as _base_identifiers


@model.over("identifiers", "^035__", override_tag=True)
@for_each_value
def identifiers(self, key, value):
"""Translates 035__ identifiers.

- scheme empty or 'CERN ISOLDE': stored as related_identifier with scheme
'other' and identifier pattern 'ISOLDE:<a>', relation 'references'.
- everything else (incl. CERCER → aleph): delegated to the base rule.
"""
scheme = value.get("9", "").strip()
system_control_number = value.get("a", "").strip()

if not scheme or scheme.upper() == "CERN ISOLDE":
related_identifiers = self.get("related_identifiers", [])
new_id = {
"identifier": f"ISOLDE:{system_control_number}",
"scheme": "other",
"relation_type": {"id": "references"},
}
if new_id not in related_identifiers:
related_identifiers.append(new_id)
self["related_identifiers"] = related_identifiers
raise IgnoreKey("identifiers")

return _base_identifiers.__wrapped__(self, key, value)


@model.over("medium", "^340__")
@for_each_value
def medium(self, key, value):
"""Ignores 340__a when value is 'paper', raises for anything else."""
a_value = value.get("a", "").strip().lower()
if a_value == "paper":
raise IgnoreKey("medium")
raise UnexpectedValue(field=key, subfield="a", value=value, stage="transform")


@model.over("additional_descriptions", "^852__")

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

isn't it easier to add these fields to ignored? also, by raising IgnoreKey at the end you might also silently ignore other subfields of this field

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

hmm but I want to ignore specific values of the 852 subfields only. Keeping in mind also that this could go to the research/base class. Isn't it better like that?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

in the ignore keys you can only add the specific ones, you can't add the whole field
but I am re-reading this and now I understood why you do this, sorry 😅 LGTM

@for_each_value
def holdings(self, key, value):
"""852 holdings: selectively ignores known location values.

- 852__c == 'CERN ARC Library': ignored.
- 852__h (depot code): ignored; records will be bulk-updated post-migration
once the ATOM/CLC target mapping is established.
- anything else: raises UnexpectedValue to surface unknown cases.
"""
h_value = StringValue(value.get("h", "")).parse()
c_value = StringValue(value.get("c", "")).parse()

# Depot location: check the value exists, then ignore.
# Records will be bulk-updated post-migration via ATOM/CLC mapping.
if h_value:
raise IgnoreKey("additional_descriptions")

if c_value and c_value.lower() == "cern arc library":
raise IgnoreKey("additional_descriptions")

if h_value and "cern depot" in h_value.lower():
raise IgnoreKey("additional_descriptions")

if c_value:
raise UnexpectedValue(field=key, subfield="c", value=value, stage="transform")

if h_value:
raise UnexpectedValue(field=key, subfield="h", value=value, stage="transform")

raise IgnoreKey("additional_descriptions")
1 change: 1 addition & 0 deletions setup.cfg
Original file line number Diff line number Diff line change
Expand Up @@ -237,6 +237,7 @@ cds_migrator_kit.migrator.rules.isolde =
base = cds_migrator_kit.transform.xml_processing.rules.base
base_records = cds_migrator_kit.rdm.records.transform.xml_processing.rules.base
publication = cds_migrator_kit.rdm.records.transform.xml_processing.rules.research
isolde = cds_migrator_kit.rdm.records.transform.xml_processing.rules.isolde
cds_migrator_kit.migrator.rules.north_area =
base = cds_migrator_kit.transform.xml_processing.rules.base
base_records = cds_migrator_kit.rdm.records.transform.xml_processing.rules.base
Expand Down
Loading