diff --git a/docs/adr/hermes-prov-diagram/hermes-prov.drawio b/docs/adr/hermes-prov-diagram/hermes-prov.drawio new file mode 100644 index 00000000..6c27a07e --- /dev/null +++ b/docs/adr/hermes-prov-diagram/hermes-prov.drawio @@ -0,0 +1,5134 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/docs/adr/hermes-prov-diagram/hermes-prov.drawio.license b/docs/adr/hermes-prov-diagram/hermes-prov.drawio.license new file mode 100644 index 00000000..e4d2c6e9 --- /dev/null +++ b/docs/adr/hermes-prov-diagram/hermes-prov.drawio.license @@ -0,0 +1,3 @@ +SPDX-FileCopyrightText: 2026 German Aerospace Center (DLR) + +SPDX-License-Identifier: CC-BY-SA-4.0 \ No newline at end of file diff --git a/docs/adr/hermes-prov-diagram/hermes-prov.svg b/docs/adr/hermes-prov-diagram/hermes-prov.svg new file mode 100644 index 00000000..aea20661 --- /dev/null +++ b/docs/adr/hermes-prov-diagram/hermes-prov.svg @@ -0,0 +1,4 @@ + + + +wasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAssociatedWithusedusedwasGeneratedBywasAttributedToactedOnBehalfOfwasDerivedFromwasInfluencedBywasDerivedFromwasGeneratedBywasAssociatedWithwasAssociatedWithusedwasDerivedFromwasInfluencedBywasDerivedFromwasGeneratedBywasAssociatedWithusedwasDerivedFromwasGeneratedBywasAssociatedWithusedwasDerivedFromwasGeneratedBywasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasAttributedTousedwasDerivedFromwasGeneratedByusedwasDerivedFromwasGeneratedBywasAssociatedWithusedwasDerivedFromwasGeneratedBywasDerivedFromwasDerivedFromwasDerivedFromwasAssociatedWithwasAssociatedWithwasInfluencedBywasAssociatedWithwasAttributedTowasAttributedTowasAttributedTowasAssociatedWithwasGeneratedByusedusedwasDerivedFromwasDerivedFromwasGeneratedBywasAssociatedWithwasAttributedTowasAssociatedWithwasAssociatedWithwasAssociatedWithwasInfluencedByusedwasDerivedFromactedOnBehalfOfusedusedusedwasDerivedFromwasDerivedFromwasGeneratedByactedOnBehalfOfwasAttributedTowasAssociatedWithactedOnBehalfOfwasDerivedFromwasInfluencedBywasAttributedTowasAttributedTowasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasGeneratedBywasGeneratedBywasGeneratedBywasAssociatedWithwasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAttributedTowasAssociatedWithwasAttributedTowasAttributedTousedusedwasAssociatedWithwasGeneratedBywasAssociatedWithwasAttributedTowasAssociatedWithwasInformedBywasAttributedTowasGeneratedBywasInformedBywasGeneratedByusedusedwasGeneratedBywasInformedByusedwasDerivedFromwasDerivedFromactedOnBehalfOfwasAttributedTowasAssociatedWithusedusedusedusedusedwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromusedusedwasAssociatedWithwasInformedByusedusedwasInformedBywasGeneratedBywasInformedByusedwasDerivedFromwasGeneratedByusedwasInformedBywasDerivedFromwasGeneratedBywasDerivedFromusedwasGeneratedBywasInformedByusedwasGeneratedBywasGeneratedByusedusedusedusedwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAssociatedWithwasInformedByusedusedusedusedusedwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromusedwasAssociatedWithwasInformedByusedwasInformedBywasGeneratedBywasInformedByusedwasDerivedFromwasGeneratedByusedwasInformedBywasDerivedFromwasGeneratedBywasDerivedFromusedwasGeneratedBywasInformedByusedwasGeneratedBywasGeneratedByusedusedusedwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAssociatedWithwasAssociatedWithwasAttributedTowasAttributedTowasAssociatedWithwasAttributedTowasInformedBywasInformedBywasInformedBywasInformedBywasDerivedFromwasAssociatedWithwasInformedBywasAttributedTowasGeneratedByusedwasDerivedFromwasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasInformedBywasAttributedTowasAssociatedWithwasAttributedTowasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromusedwasGeneratedBywasGeneratedBywasGeneratedBywasAssociatedWithwasAssociatedWithactedOnBehalfOfactedOnBehalfOfwasGeneratedBywasGeneratedByactedOnBehalfOfwasGeneratedBywasAttributedTowasAssociatedWithactedOnBehalfOfusedwasAssociatedWithwasAttributedTowasAttributedTowasAttributedTowasAttributedTousedwasAssociatedWithwasAttributedTowasAttributedTowasAttributedTowasAttributedTowasAttributedTousedusedusedwasAssociatedWithusedusedwasAssociatedWithwasInformedBywasInformedBywasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasGeneratedBywasGeneratedByusedusedusedwasAttributedTowasAssociatedWithwasDerivedFromwasGeneratedBywasDerivedFromwasGeneratedBywasGeneratedBywasAssociatedWithwasAssociatedWithactedOnBehalfOfwasAssociatedWithwasAssociatedWithwasAssociatedWithwasAssociatedWithactedOnBehalfOfwasAssociatedWithwasInformedBywasInformedBywasAssociatedWithwasAttributedTowasDerivedFromwasDerivedFromwasDerivedFromwasGeneratedBywasGeneratedByusedusedusedwasAttributedTowasAssociatedWithwasDerivedFromwasGeneratedBywasDerivedFromwasGeneratedBywasGeneratedBywasAssociatedWithwasInformedBywasInformedBywasDerivedFromwasDerivedFromwasDerivedFromwasDerivedFromactedOnBehalfOfactedOnBehalfOfwasAttributedTowasDerivedFromwasGeneratedBywasGeneratedByusedusedusedwasAttributedTowasAssociatedWithwasGeneratedBywasGeneratedBywasGeneratedBywasAssociatedWithwasAssociatedWithwasAssociatedWithwasAssociatedWithactedOnBehalfOfactedOnBehalfOfharvest pluginname, version, settingsharvest sourcepath uri.hermes/harvest/{plugin_name}/codemeta.jsontext, path uri, date createdharvested metadatadatasoftware-metadatadate, timemapend timewritestart time, end time.hermes/harvest/{plugin_name}/expanded.jsontext, path uri, date created.hermes/harvest/{plugin_name}/context.jsontext, path uri, date createdLegenddesignmeaningprovenance: Agentprovenance: Entityprovenance: Activitybold textrecord those properties alwayssolid liningrecord as detailed as possibledashed liningrecord without many detailsgrayed outoptional / not always existentnamepropertiesnamepropertiesnamepropertiesharvest pluginname, version, settingsharvest sourcepath uri.hermes/harvest/{plugin_name}/codemeta.jsontext, path uri, date createdharvested metadatadatasoftware-metadatadata, timemapend timewritestart time, end time.hermes/harvest/{plugin_name}/expanded.jsontext, path uri, date createdHARVESThermesversionHERMES cacheloadfunc, args, kwargs, source, timeharvest base pluginsettingsloadfunc, args, kwargs, source, timeharvest commandsettings.hermes/harvest/{plugin_name}/context.jsontext, path uri, date createdprocess pluginname, version, settingsmerge strategiesstrategies, timeprocess pluginname, version, settings.hermes/process/result/codemeta.jsontext, path uri, date createdmerge strategiesstrategies, timemerge strategiesstart time, end timewritestart time, end time.hermes/process/result/expanded.jsontext, path uri, date createdPROCESSgenerate merge strategiesstart time, end timeprocess base pluginsettingsgenerate merge strategiesstart time, end timeprocess commandsettings.hermes/process/result/context.jsontext, path uri, date createdprocess pluginname, version, settingsgenerate merge strategiesstart time, end timemerge strategiesstart time, end timemerged strategiesstrategies, timemerged strategiesstrategies, timeharvest pluginname, version, settingsharvest sourcepath uri.hermes/harvest/{plugin_name}/codemeta.jsontext, path uri, date createdharvested metadatadatasoftware-metadatadata, timemapend timewritestart time, end time.hermes/harvest/{plugin_name}/expanded.jsontext, path uri, date created.hermes/harvest/{plugin_name}/context.jsontext, path uri, date createdloadfunc, args, kwargs, source, timesoftware-metadatadata, timeloadstart time, end timesoftware-metadatadata, timeloadstart time, end timesoftware-metadatadata, timeloadstart time, end timeusedusedwasAssociatedWithreject/ replace/ ...value with other valuestart time, end time, strategy usedmerge value at keykey, strategy usedmergestart time, end timesoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedsoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedsoftware-metadatatime, datasoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedreject/ replace/ ...value with other valuestart time, end time, strategy usedmerge value at keykey, strategy usedmergestart time, end timesoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedsoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedsoftware-metadatatime, datasoftware-metadatatime, datareject/ replace/ ...value with other valuestart time, end time, strategy usedmerge strategiesstrategies, timecurate commandsettingssoftware-metadatadata, timeloadstart time, end timecurate base pluginsettingscurate pluginname, version, settingssoftware-metadatadata, time.hermes/curate/result/codemeta.jsontext, path uri, time createdwritestart time, end time.hermes/curate/result/expanded.jsontext, path uri, time created.hermes/curate/result/context.jsontext, path uri, time createdCURATEusedwasDerivedFromactedOnBehalfOfusedusedusedwasDerivedFromwasDerivedFromwasGeneratedByactedOnBehalfOfwasAttributedTowasAssociatedWithwasAssociatedWithactedOnBehalfOfwasDerivedFromwasAttributedTowasAttributedTowasDerivedFromwasGeneratedBywasAssociatedWithwasAssociatedWithwasAssociatedWithwasAssociatedWithdeposit commandsettingssoftware-metadatadata, timeloadstart time, end timedeposit base pluginsettingsdeposit pluginname, version, settingsmapped data for depositdata, time.hermes/deposit/{deposit_plugin}/deposit.jsontext, path uri, time createdwritestart time, end timeDEPOSITmapstart time, end timeupdated metadatadata, time.hermes/deposit/{deposit_plugin}/result.jsontext, path uri, time createdwritestart time, end timeusedwasDerivedFromactedOnBehalfOfusedwasGeneratedByactedOnBehalfOfwasAttributedTowasAssociatedWithwasAssociatedWithactedOnBehalfOfwasDerivedFromwasInfluencedBywasDerivedFromwasGeneratedBywasAssociatedWithwasAssociatedWithwasAssociatedWithpostprocess commandsettingsdeposit resultdata, timeloadstart time, end timepostprocess base pluginsettingspostprocess pluginname, version, settingsprocessed datadata, timesome datatext, path uri, time createdwritestart time, end timePOSTPROCESSwasAttributedToprocessed datadata, timesome datatext, path uri, time createdwritestart time, end timeloaded datadata, timesome datapath uriloadstart time, end timeloaded datadata, timesome datapath uriloadstart time, end timewasAttributedTowasAssociatedWithwasAssociatedWithwasAssociatedWithdeposit resultdata, timeloadstart time, end timepostprocess pluginname, version, settingsprocessed datadata, timesome datatext, path uri, time createdwritestart time, end timewasAttributedToprocessed datadata, timesome datatext, path uri, time createdwritestart time, end timeloaded datadata, timesome datapath uriloadstart time, end timeloaded datadata, timesome datapath uriloadstart time, end timewasAttributedTo \ No newline at end of file diff --git a/docs/adr/hermes-prov-diagram/hermes-prov.svg.license b/docs/adr/hermes-prov-diagram/hermes-prov.svg.license new file mode 100644 index 00000000..e4d2c6e9 --- /dev/null +++ b/docs/adr/hermes-prov-diagram/hermes-prov.svg.license @@ -0,0 +1,3 @@ +SPDX-FileCopyrightText: 2026 German Aerospace Center (DLR) + +SPDX-License-Identifier: CC-BY-SA-4.0 \ No newline at end of file diff --git a/docs/source/adr/0012-overall-data-model-design.md b/docs/source/adr/0012-overall-data-model-design.md index 2347f532..5aae2867 100644 --- a/docs/source/adr/0012-overall-data-model-design.md +++ b/docs/source/adr/0012-overall-data-model-design.md @@ -22,13 +22,13 @@ Superseded: we no longer need to serialize additional information like provenanc ## Considered Options * One common model for all stages -* Seperate model for different stages +* Separate model for different stages * Common model for all stages * Processing model and curated model ## Decision Outcome -Chosen option: "Seperate model for different stages", because comes out best. +Chosen option: "Separate model for different stages", because comes out best. ## Pros and Cons of the Options diff --git a/docs/source/conf.py b/docs/source/conf.py index c627daba..fd30df54 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -28,6 +28,7 @@ sys.path.insert(0, os.path.abspath('../../src')) sys.path.append(os.path.abspath('_ext')) + def read_from_pyproject(file_path="../../pyproject.toml"): """ Reads the metadata from the pyproject.toml file. @@ -51,6 +52,7 @@ def read_from_pyproject(file_path="../../pyproject.toml"): except Exception as e: return f"An unexpected error occurred: {e}" + def read_authors_from_pyproject(): metadata = read_from_pyproject() authors = metadata.get("authors", []) @@ -59,6 +61,7 @@ def read_authors_from_pyproject(): # Convert the list of authors to a comma-separated string return ", ".join([author["name"] for author in authors]) + def read_version_from_pyproject(): metadata = read_from_pyproject() version = metadata.get("version", "") @@ -70,7 +73,8 @@ def read_version_from_pyproject(): # -- Project information ----------------------------------------------------- project = 'HERMES Workflow' -copyright = '2025 by Forschungszentrum Jülich (FZJ), German Aerospace Center (DLR) and Helmholtz-Zentrum Dresden-Rossendorf (HZDR)' +copyright = '2025 by Forschungszentrum Jülich (FZJ), German Aerospace Center (DLR)' \ + ' and Helmholtz-Zentrum Dresden-Rossendorf (HZDR)' author = read_authors_from_pyproject() # The full version, including alpha/beta/rc tags @@ -191,6 +195,7 @@ def read_version_from_pyproject(): # TODO: remove this workaround and remove "undoc-members" from autoapi_options once everything is documented # This removes all generated entries for known documented classes (because autoapi will add all attributes # it finds in the code no matter if they are described in a class doc string or not). + def autoapi_skip_member(app, obj_type, name, obj, skip, options): if obj_type == "attribute": if any(documented_type in obj.id for documented_type in [ @@ -201,5 +206,6 @@ def autoapi_skip_member(app, obj_type, name, obj, skip, options): return skip + def setup(app): app.connect("autoapi-skip-member", autoapi_skip_member) diff --git a/hermes.toml b/hermes.toml index a42a9406..dab72523 100644 --- a/hermes.toml +++ b/hermes.toml @@ -5,6 +5,9 @@ [harvest] sources = [ "cff", "toml" ] # ordered priority (first one is most important) +[process] +plugins = [ "invenio", "codemeta" ] + [curate] plugin = "pass_curate" diff --git a/pyproject.toml b/pyproject.toml index 23c28b63..6b62aa65 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -69,11 +69,14 @@ rodare = "hermes.commands.deposit.rodare:RodareDepositPlugin" [project.entry-points."hermes.postprocess"] config_invenio_record_id = "hermes.commands.postprocess.invenio:config_record_id" config_invenio_rdm_record_id = "hermes.commands.postprocess.invenio_rdm:config_record_id" -cff_doi = "hermes.commands.postprocess.invenio:cff_doi" -codemeta_doi = "hermes.commands.postprocess.invenio:codemeta_doi" +invenio_cff_doi = "hermes.commands.postprocess.invenio:cff_doi" +invenio_rdm_cff_doi = "hermes.commands.postprocess.invenio_rdm:cff_doi" +invenio_codemeta_doi = "hermes.commands.postprocess.invenio:codemeta_doi" +invenio_rdm_codemeta_doi = "hermes.commands.postprocess.invenio_rdm:codemeta_doi" [project.entry-points."hermes.process"] codemeta = "hermes.commands.process.standard_merge:CodemetaProcessPlugin" +invenio = "hermes.commands.process.invenio_merge:InvenioProcessPlugin" [project.entry-points."hermes.curate"] pass_curate = "hermes.commands.curate.pass_curate:PassCuratePlugin" diff --git a/src/hermes/commands/__init__.py b/src/hermes/commands/__init__.py index 5203ac18..2733d02e 100644 --- a/src/hermes/commands/__init__.py +++ b/src/hermes/commands/__init__.py @@ -17,3 +17,4 @@ from hermes.commands.process.base import HermesProcessCommand from hermes.commands.deposit.base import HermesDepositCommand from hermes.commands.postprocess.base import HermesPostprocessCommand +from hermes.commands.report.base import HermesReportCommand diff --git a/src/hermes/commands/base.py b/src/hermes/commands/base.py index fd24cd22..889eaf67 100644 --- a/src/hermes/commands/base.py +++ b/src/hermes/commands/base.py @@ -9,7 +9,8 @@ import logging import pathlib from importlib import metadata -from typing import Type, Union +from typing import Optional +from typing_extensions import Self from pydantic import BaseModel from pydantic_settings import BaseSettings, SettingsConfigDict @@ -17,36 +18,50 @@ class HermesSettings(BaseSettings): - """Root class for HERMES configuration model.""" + """ + Root class for HERMES configuration model. - model_config = SettingsConfigDict(env_file_encoding='utf-8') + Attributes: + model_config (SettingsConfigDict): The settings config dict for the settings of hermes. + """ - logging: dict = {} + model_config = SettingsConfigDict(env_file_encoding='utf-8') class HermesCommand(abc.ABC): """Base class for a HERMES workflow command. - :cvar NAME: The name of the sub-command that is defined here. + Attributes: + command_name (str): (class attribute) Only defined here for highlighting, the value of the subclass is is used. + settings_class (type): (class attribute) The settings class for the general hermes command settings """ command_name: str = "" - settings_class: Type = HermesSettings + settings_class: type = HermesSettings - def __init__(self, parser: argparse.ArgumentParser): - """Initialize a new instance of any HERMES command. + def __init__(self: Self, parser: argparse.ArgumentParser) -> None: + """ + Initialize a new instance of any HERMES command. + + Args: + parser (ArgumentParser): The command line parser used for reading command line arguments. - :param parser: The command line parser used for reading command line arguments. + Returns: + None: """ self.parser = parser self.plugins = self.init_plugins() self.settings = None self.log = logging.getLogger(f"hermes.{self.command_name}") - self.errors = [] - def init_plugins(self): - """Collect and initialize the plugins available for the HERMES command.""" + def init_plugins(self: Self) -> dict[str, type["HermesPlugin"]]: + """ + Collect and initialize the plugins available for the HERMES command. + + Returns: + dict[str, HermesPlugin]: A map mapping the plugin name to the plugin class for the current step. + """ # Collect all entry points for this group (i.e., all valid plug-ins for the step) entry_point_group = f"hermes.{self.command_name}" @@ -65,10 +80,20 @@ def init_plugins(self): return group_plugins @classmethod - def derive_settings_class(cls, setting_types: dict[str, Type]) -> None: - """Build a new Pydantic data model class for configuration. + def derive_settings_class(cls: type[Self], setting_types: dict[str, type["HermesPlugin"]]) -> None: + """ + Build a new Pydantic data model class for configuration. This will create a new class that includes all settings from the plugins available. + + Args: + settings_types (dict[str, type]): The settings classes for the plugins. + + Returns: + None: + + Raises: + ValueError: If the command has no settings. """ if cls.settings_class is not None: @@ -88,10 +113,16 @@ def derive_settings_class(cls, setting_types: dict[str, Type]) -> None: elif setting_types: raise ValueError(f"Command {cls.command_name} has no settings, hence plugin must not have settings, too.") - def init_common_parser(self, parser: argparse.ArgumentParser) -> None: - """Initialize the common command line arguments available for all HERMES sub-commands. + def init_common_parser(self: Self, parser: argparse.ArgumentParser) -> None: + """ + Initialize the common command line arguments available for all HERMES sub-commands. - :param parser: The base command line parser used as entry point when reading command line arguments. + Args: + parser (ArgumentsParser): The base command line parser used as entry point when reading command line + arguments. + + Returns: + None: """ parser.add_argument( @@ -117,26 +148,47 @@ def init_common_parser(self, parser: argparse.ArgumentParser) -> None: "VALUE is the actual value.", ) - def init_command_parser(self, command_parser: argparse.ArgumentParser) -> None: - """Initialize the command line arguments available for this specific HERMES sub-commands. + def init_command_parser(self: Self, command_parser: argparse.ArgumentParser) -> None: + """ + Initialize the command line arguments available for this specific HERMES sub-commands. You should override this method to add your custom arguments to the command line parser of the respective sub-command. - :param command_parser: The command line sub-parser responsible for the HERMES sub-command. + Args: + command_parser (ArgumentParser): The command line sub-parser responsible for the HERMES sub-command. + + Returns: + None: """ pass - def load_settings(self, args: argparse.Namespace): - """Load settings from the configuration file (passed in from command line).""" + def load_settings(self: Self, args: argparse.Namespace) -> None: + """ + Load settings from the configuration file (passed in from command line). + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ toml_data = tomlkit.load((args.path / args.config).open()).unwrap() self.root_settings = HermesCommand.settings_class.model_validate(toml_data) self.settings = getattr(self.root_settings, self.command_name) - def patch_settings(self, args: argparse.Namespace): - """Process command line options for the settings.""" + def patch_settings(self: Self, args: argparse.Namespace) -> None: + """ + Process command line options for the settings. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ for key, value in args.options: target = self.settings @@ -148,27 +200,42 @@ def patch_settings(self, args: argparse.Namespace): setattr(target, sub_keys[-1], value) @abc.abstractmethod - def __call__(self, args: argparse.Namespace): + def __call__(self: Self, args: argparse.Namespace) -> None: """Execute the HERMES sub-command. - :param args: The namespace that was returned by the command line parser when reading the arguments. + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: """ pass class HermesPlugin(abc.ABC): - """Base class for all HERMES plugins.""" + """ + Base class for all HERMES plugins. + + Attributes: + plugin_node: ... + settings_class: The settings_class of the plugin. + """ pluing_node = None - settings_class: Union[Type, None] = None + settings_class: Optional[type] = None @abc.abstractmethod - def __call__(self, command: HermesCommand) -> None: - """Execute the plugin. + def __call__(self: Self, command: HermesCommand) -> None: + """ + Execute the plugin. + + Args: + command (HermesCommand): The command that triggered this plugin to run. - :param command: The command that triggered this plugin to run. + Returns: + None: """ pass @@ -180,12 +247,27 @@ class HermesHelpSettings(BaseModel): class HermesHelpCommand(HermesCommand): - """Show help page and exit.""" + """ + Show help page and exit. + + Attributes: + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general help settings. + """ command_name = "help" settings_class = HermesHelpSettings - def init_command_parser(self, command_parser: argparse.ArgumentParser) -> None: + def init_command_parser(self: Self, command_parser: argparse.ArgumentParser) -> None: + """ + Add arguments for help command. + + Args: + command_parser (ArgumentParser): The used argument parser. + + Returns: + None: + """ command_parser.add_argument( "subcommand", nargs="?", @@ -193,7 +275,16 @@ def init_command_parser(self, command_parser: argparse.ArgumentParser) -> None: help="The HERMES sub-command to get help for.", ) - def __call__(self, args: argparse.Namespace) -> None: + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ if args.subcommand: # When a sub-command is given, show its help page (i.e., by "running" the command with "-h" flag). self.parser.parse_args([args.subcommand, "-h"]) @@ -209,15 +300,38 @@ class HermesVersionSettings(BaseModel): class HermesVersionCommand(HermesCommand): - """Show HERMES version and exit.""" + """ + Show HERMES version and exit. + + Attributes: + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general help settings. + """ command_name = "version" settings_class = HermesVersionSettings - def load_settings(self, args: argparse.Namespace): - """Pass loading settings as not necessary for this command.""" + def load_settings(self: Self, args: argparse.Namespace) -> None: + """ + Pass loading settings as not necessary for this command. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ pass - def __call__(self, args: argparse.Namespace) -> None: + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ self.log.info(metadata.version("hermes")) self.parser.exit() diff --git a/src/hermes/commands/cli.py b/src/hermes/commands/cli.py index 68cc23e1..cbc645e0 100644 --- a/src/hermes/commands/cli.py +++ b/src/hermes/commands/cli.py @@ -14,7 +14,7 @@ from hermes import logger from hermes.commands import ( HermesCurateCommand, HermesCleanCommand, HermesDepositCommand, HermesHarvestCommand, HermesHelpCommand, - HermesInitCommand, HermesPostprocessCommand, HermesProcessCommand, HermesVersionCommand + HermesInitCommand, HermesPostprocessCommand, HermesProcessCommand, HermesReportCommand, HermesVersionCommand ) from hermes.commands.base import HermesCommand from hermes.error import HermesPluginRunError @@ -46,6 +46,7 @@ def main() -> None: HermesInitCommand(parser), HermesPostprocessCommand(parser), HermesProcessCommand(parser), + HermesReportCommand(parser), HermesVersionCommand(parser), ): if command.settings_class is not None: diff --git a/src/hermes/commands/curate/base.py b/src/hermes/commands/curate/base.py index 0661fa36..d0016c4a 100644 --- a/src/hermes/commands/curate/base.py +++ b/src/hermes/commands/curate/base.py @@ -5,6 +5,9 @@ # SPDX-FileContributor: Michael Meinel import argparse +import datetime +from typing import Optional +from typing_extensions import Self from pydantic import BaseModel @@ -13,39 +16,92 @@ from hermes.model import SoftwareMetadata from hermes.model.hermes_cache import HermesCacheManager from hermes.model.error import HermesValidationError +from hermes.model.provenance.ld_prov import ld_prov_list class HermesCuratePlugin(HermesPlugin): """ Base plugin for curate plugins. """ - def __call__(self, command: HermesCommand, metadata: SoftwareMetadata) -> SoftwareMetadata: + def __call__(self: Self, command: "HermesCurateCommand", metadata: SoftwareMetadata) -> SoftwareMetadata: + """ + Execute the hermes curate plugin `self`. + + Args: + command (HermesCurateCommand): The command being executed. + metadata (SoftwareMetadata): The metadata to be curated. + + Returns: + SoftwareMetadata: The curated metadata. + """ pass class CurateSettings(BaseModel): - """Generic deposition settings.""" + """ + Generic deposition settings. + + Attributes: + plugin (str): The plugin to be executed. + """ plugin: str = "pass_curate" class HermesCurateCommand(HermesCommand): - """ Curate the unified metadata before deposition. """ + """ + Curate the unified metadata before deposition. + + Attributes: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general curate settings. + """ + + command_name: str = "curate" + settings_class: type = CurateSettings + + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + + Raises: + HermesValidationError: If the results of the process step couldn't be loaded. + MisconfigurationError: If the curation plugin wasn't found. + HermesPluginRunError: If something went wrong in the plugin run. + """ + self.args = args + self.log.info("# Load provenance data from process step") + # try loading and adding general information to the provenance document + prov_doc = self.load_prov_doc() + if prov_doc is not None: + prov_doc.add_hermes_settings(self) + prov_doc.add_settings_to_command("curate", self) + # get basic hermes objects to reference later + curate_command = prov_doc.get_hermes_command("curate") + curate_base_plugin = prov_doc.get_hermes_base_plugin("curate") + process_command = prov_doc.get_hermes_command("process") + hermes_cache = prov_doc.get_hermes_cache() - command_name = "curate" - settings_class = CurateSettings - - def __call__(self, args: argparse.Namespace) -> None: self.log.info("# Metadata curation") plugin_name = self.settings.plugin ctx = HermesCacheManager() + ctx.prepare_step("curate") self.log.info("## Load processed metadata") # load processed data ctx.prepare_step("process") try: + begin_load_at_time = datetime.datetime.now() metadata = SoftwareMetadata.load_from_cache(ctx, "result") + end_load_at_time = datetime.datetime.now() except Exception as e: self.log.critical( "## The data from the process step could not be loaded or is invalid for some reason.", @@ -54,6 +110,9 @@ def __call__(self, args: argparse.Namespace) -> None: raise HermesValidationError("The results of the process step are invalid.") from e ctx.finalize_step("process") + # save loaded metadata now, because it could be altered in curation + loaded_metadata_str = str(metadata.compact()) + self.log.info(f"## Load curation plugin {plugin_name}") # load plugin try: @@ -66,12 +125,124 @@ def __call__(self, args: argparse.Namespace) -> None: # run plugin try: curated_metadata = plugin_func(self, metadata) + end_curation_time = datetime.datetime.now() except Exception as e: self.log.critical(f"## Unknown error while executing the {plugin_name} plugin.", exc_info=1) raise HermesPluginRunError(f"Something went wrong while running the curate plugin {plugin_name}") from e self.log.info("## Store curated data") # store metadata + begin_store_at_time = datetime.datetime.now() curated_metadata.write_to_cache(ctx, "result") + stored_at_time = datetime.datetime.now() + + if prov_doc is not None: + # add information on the curate plugin + curate_plugin = prov_doc.add_hermes_plugin("curate", plugin_name, plugin_func, self) + # get objects from process + store_action_of_process = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [process_command.ref, hermes_cache.ref] and + "prov:wasInformedBy" in node + ))[0] + stored_results_of_process = [res.ref for res in prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action_of_process.ref] + ))] + # add information on load, loaded data and curated data + load_action = prov_doc.add_activity(data={ + "schema:description": "loads the data from process step", + "prov:wasAssociatedWith": [process_command.ref, hermes_cache.ref], + "prov:used": stored_results_of_process, + "prov:startedAtTime": begin_load_at_time, + "prov:endedAtTime": end_load_at_time + }) + loaded_data = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "data loaded from process step", + "schema:text": loaded_metadata_str, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:wasGeneratedBy": load_action.ref, + "prov:wasDerivedFrom": stored_results_of_process, + "prov:generatedAtTime": end_load_at_time + }) + curated_data = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "curated metadata", + "schema:text": str(curated_metadata.compact()), + "prov:wasAttributedTo": [curate_plugin.ref, curate_base_plugin.ref, curate_command.ref], + "prov:wasInfluencedBy": curate_plugin.ref, + "prov:wasGeneratedBy": load_action.ref, + "prov:wasDerivedFrom": loaded_data.ref, + "prov:generatedAtTime": end_curation_time + }) + # add provenance information on the write and stored curated metadata + write = prov_doc.add_activity(data={ + "schema:description": "Writes the processed metadata into the HERMES cache.", + "prov:wasAssociatedWith": [curate_command.ref, hermes_cache.ref], + "prov:used": curated_data.ref, + "prov:startedAtTime": begin_store_at_time, + "prov:endedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The compacted version of the processed metadata.", + "schema:text": str(curated_metadata.compact()), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "curate" / "result" / "codemeta.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": curated_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The context of the processed metadata.", + "schema:text": str({"@context": curated_metadata.full_context}), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "curate" / "result" / "context.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": curated_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The expanded version of the processed metadata.", + "schema:text": str(curated_metadata.ld_value), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "curate" / "result" / "expanded.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": curated_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + + # store provenance information + with ctx["provenance"] as cache: + cache["result"] = prov_doc.ld_value ctx.finalize_step("curate") + + def load_prov_doc(self: Self) -> Optional[ld_prov_list]: + """ + Loads the provenance document of the process step. + + Returns: + ld_prov_list | None: The loaded provenance document or None if the load failed. + """ + # set up HermesContext + ctx = HermesCacheManager() + ctx.prepare_step("process") + with ctx["provenance"] as cache: + # try load + try: + return ld_prov_list.load_ld_prov_list(cache["result"]) + except Exception: + # log the warning and return None + self.log.warning( + "The provenance data from the process step could not be loaded. " + "Processing will proceed without collecting provenance data.", + exc_info=1 + ) + finally: + ctx.finalize_step("process") diff --git a/src/hermes/commands/deposit/base.py b/src/hermes/commands/deposit/base.py index bb23cd81..551dfda7 100644 --- a/src/hermes/commands/deposit/base.py +++ b/src/hermes/commands/deposit/base.py @@ -4,9 +4,13 @@ # SPDX-FileContributor: David Pape # SPDX-FileContributor: Michael Meinel +# SPDX-FileContributor: Michael Fritzsche import abc import argparse +import datetime +from typing import Optional +from typing_extensions import Self from pydantic import BaseModel @@ -15,129 +19,346 @@ from hermes.model.hermes_cache import HermesCacheManager from hermes.model import SoftwareMetadata from hermes.model.error import HermesValidationError +from hermes.model.provenance.ld_prov import ld_prov_list class BaseDepositPlugin(HermesPlugin): - """Base class that implements the generic deposition workflow. + """ + Base class that implements the generic deposition workflow. + + Attributes: + command (HermesCommand): The command running this plugin. + metadata (SoftwareMetadata): The loaded curated metadata. TODO: describe workflow... needs refactoring to be less stateful! """ - def __call__(self, command: HermesCommand) -> None: - """Initiate the deposition process. + def __call__(self: Self, command: "HermesDepositCommand", prov_doc: Optional[ld_prov_list]) -> None: + """ + Initiate the deposition process. This calls a list of additional methods on the class, none of which need to be implemented. + + Args: + command (HermesDepositCommand): The command running this plugin. + prov_doc (ld_prov_list | None): The provenance document the provenance information is to be recorded in. + + Returns: + None: + + Raises: + HermesValidationError: If the metadata from the curation step couldn't be loaded. """ self.command = command - self.ctx = HermesCacheManager() - self.ctx.prepare_step("deposit") + target = command.settings.target + + ctx = HermesCacheManager() - self.ctx.prepare_step("curate") + ctx.prepare_step("curate") + # load curated metadata try: - self.metadata = SoftwareMetadata.load_from_cache(self.ctx, "result") + start_of_load = datetime.datetime.now() + self.metadata = SoftwareMetadata.load_from_cache(ctx, "result") + end_of_load = datetime.datetime.now() except Exception as e: raise HermesValidationError("The results of the curate step are invalid.") from e - self.ctx.finalize_step("curate") - + ctx.finalize_step("curate") + + if prov_doc is not None: + # add provenance information on the plugin + plugin = prov_doc.add_hermes_plugin("deposit", target, self, command) + # get basic hermes objects to reference later + deposit_command = prov_doc.get_hermes_command("deposit") + curate_command = prov_doc.get_hermes_command("curate") + deposit_base_plugin = prov_doc.get_hermes_base_plugin("deposit") + hermes_cache = prov_doc.get_hermes_cache() + # get objects from the curate step + store_action_curate = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [curate_command.ref, hermes_cache.ref] and + "prov:used" in node and + len(node["prov:used"]) == 1 + ))[0] + results_curate = [item.ref for item in prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action_curate.ref] + ))] + # record provenance information on the load action and the loaded data + load_action = prov_doc.add_activity(data={ + "schema:description": "Loads the results of the curate step.", + "prov:used": results_curate, + "prov:wasAssociatedWith": [hermes_cache.ref, deposit_command.ref, deposit_base_plugin.ref], + "prov:startedAtTime": start_of_load, + "prov:endedAtTime": end_of_load + }) + loaded_data = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "data loaded from curate step", + "schema:text": str(self.metadata.compact()), + "prov:wasAttributedTo": hermes_cache.ref, + "prov:wasGeneratedBy": load_action.ref, + "prov:wasDerivedFrom": results_curate, + "prov:generatedAtTime": end_of_load + }) + + # prepare, map metadata and store the result self.prepare() + start_of_map = datetime.datetime.now() deposit = self.map_metadata() - with self.ctx[command.settings.target] as cache: + end_of_map = datetime.datetime.now() + ctx.prepare_step("deposit") + with ctx[target] as cache: cache["deposit"] = deposit - + end_of_store = datetime.datetime.now() + + if prov_doc is not None: + # record provenance information on map, mapped data, store and the stored data + map_action = prov_doc.add_activity(data={ + "schema:description": "Maps the metadata to the format required by the deposition target.", + "prov:used": loaded_data.ref, + "prov:wasAssociatedWith": plugin.ref, + "prov:startedAtTime": start_of_map, + "prov:endedAtTime": end_of_map + }) + mapped_data = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The metadata mapped to the format required by the deposition target.", + "schema:text": str(deposit), + "prov:wasAttributedTo": plugin.ref, + "prov:wasGeneratedBy": map_action.ref, + "prov:wasDerivedFrom": loaded_data.ref, + "prov:generatedAtTime": end_of_load + }) + store_mapped_data = prov_doc.add_activity(data={ + "schema:description": "Stores the mapped metadata.", + "prov:used": mapped_data.ref, + "prov:wasAssociatedWith": [hermes_cache.ref, deposit_command.ref, deposit_base_plugin.ref], + "prov:startedAtTime": end_of_map, + "prov:endedAtTime": end_of_store + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The stored version of the mapped metadata.", + "schema:text": str(deposit), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "deposit" / target / "deposit.json").absolute().as_uri(), + "prov:wasGeneratedBy": store_mapped_data.ref, + "prov:wasDerivedFrom": mapped_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": end_of_store + }) + + # create version if self.is_initial_publication(): self.create_initial_version() else: self.create_new_version() - deposit = self.update_metadata() - with self.ctx[command.settings.target] as cache: - cache["result"] = deposit - self.ctx.finalize_step("deposit") + # update mapped data and store the result + updated_deposit = self.update_metadata() + end_of_update_map = datetime.datetime.now() + with ctx[target] as cache: + cache["result"] = updated_deposit + end_of_second_store = datetime.datetime.now() + ctx.finalize_step("deposit") + + if prov_doc is not None: + # record update, store and stored data + updated_mapped_data = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The updated mapped metadata.", + "schema:text": str(updated_deposit), + "prov:wasInfluencedBy": plugin.ref, + "prov:wasDerivedFrom": mapped_data.ref, + "prov:generatedAtTime": end_of_update_map + }) + store_updated_mapped_data = prov_doc.add_activity(data={ + "schema:description": "Stores the mapped metadata.", + "prov:used": updated_mapped_data.ref, + "prov:wasAssociatedWith": [hermes_cache.ref, deposit_command.ref, deposit_base_plugin.ref], + "prov:startedAtTime": end_of_update_map, + "prov:endedAtTime": end_of_second_store + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The stored version of the updated mapped metadata.", + "schema:text": str(updated_deposit), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "deposit" / target / "result.json").absolute().as_uri(), + "prov:wasGeneratedBy": store_updated_mapped_data.ref, + "prov:wasDerivedFrom": updated_mapped_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": end_of_second_store + }) + + # finish up deposit self.delete_artifacts() self.upload_artifacts() self.publish() - def prepare(self) -> None: - """Prepare the deposition. + def prepare(self: Self) -> None: + """ + Prepare the deposition. This method may be implemented to check whether config and context match some initial conditions. If no exceptions are raised, execution continues. + + Returns: + None: """ pass @abc.abstractmethod - def map_metadata(self) -> dict: - """Map the given metadata to the target schema of the deposition platform and return it. + def map_metadata(self: Self) -> dict: + """ + Map the given metadata to the target schema of the deposition platform and return it. When mapping metadata, make sure to add traces to the HERMES software, e.g. via DataCite's ``relatedIdentifier`` using the ``isCompiledBy`` relation. Ideally, the value of the relation target should be of the respective type for DOIs in your metadata schema, with the value itself being the DOI for the version of the HERMES software you are using. + + Returns: + dict: The mapped metadata. """ pass - def is_initial_publication(self) -> bool: - """Decide whether to do an initial publication or publish a new version. + def is_initial_publication(self: Self) -> bool: + """ + Decide whether to do an initial publication or publish a new version. Returning ``True`` indicates that publication of an initial version will be executed, resulting in a call of :meth:`create_initial_version`. ``False`` indicates a new version of an existing publication, leading to a call of :meth:`create_new_version`. By default, this returns ``True``. + + Returns: + bool: Whether or not it is the initial publication. """ return True - def create_initial_version(self) -> None: - """Create an initial version of the publication on the target platform.""" + def create_initial_version(self: Self) -> None: + """ + Create an initial version of the publication on the target platform. + + Returns: + None: + """ pass - def create_new_version(self) -> None: - """Create a new version of an existing publication on the target platform.""" + def create_new_version(self: Self) -> None: + """ + Create a new version of an existing publication on the target platform. + + Returns: + None: + """ pass @abc.abstractmethod - def update_metadata(self) -> dict: - """Update the metadata of the newly created version and return it even if it hasn't changed.""" + def update_metadata(self: Self) -> dict: + """ + Update the metadata of the newly created version and return it even if it hasn't changed. + + Returns: + dict: The updated metadata. + """ pass - def delete_artifacts(self) -> None: - """Delete any superfluous artifacts taken from the previous version of the publication.""" + def delete_artifacts(self: Self) -> None: + """ + Delete any superfluous artifacts taken from the previous version of the publication. + + Returns: + None: + """ pass - def upload_artifacts(self) -> None: - """Upload new artifacts to the target platform.""" + def upload_artifacts(self: Self) -> None: + """ + Upload new artifacts to the target platform. + + Returns: + None: + """ pass @abc.abstractmethod - def publish(self) -> None: - """Publish the newly created deposit on the target platform.""" + def publish(self: Self) -> None: + """ + Publish the newly created deposit on the target platform. + + Returns: + None: + """ pass class DepositSettings(BaseModel): - """Generic deposition settings.""" + """ + Generic deposition settings. + + Attributes: + target (str): The plugin to be executed. + """ target: str = "" class HermesDepositCommand(HermesCommand): - """ Deposit the curated metadata to repositories. """ + """ + Deposit the curated metadata to repositories. + + Attributes: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general deposit settings. + """ command_name = "deposit" settings_class = DepositSettings - def init_command_parser(self, command_parser: argparse.ArgumentParser) -> None: - command_parser.add_argument('--file', '-f', nargs=1, action='append', - help="File that should be part of the deposition.") - command_parser.add_argument('--initial', action='store_true', default=False, - help="Allow initial deposition (i.e., minting a new PID).") + def init_command_parser(self: Self, command_parser: argparse.ArgumentParser) -> None: + """ + Add arguments for deposit command. + + Args: + command_parser (ArgumentParser): The used argument parser. + + Returns: + None: + """ + command_parser.add_argument( + '--file', '-f', nargs=1, action='append', help="File that should be part of the deposition." + ) + command_parser.add_argument( + '--initial', action='store_true', default=False, help="Allow initial deposition (i.e., minting a new PID)." + ) + + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. - def __call__(self, args: argparse.Namespace) -> None: + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + + Raises: + MisconfigurationError: If the deposit plugin wasn't found. + HermesPluginRunError: If something went wrong in the plugin run. + """ self.log.info("# Metadata deposition") self.args = args plugin_name = self.settings.target + # try loading and adding general information to the provenance document + prov_doc = self.load_prov_doc() + if prov_doc is not None: + prov_doc.add_hermes_settings(self) + prov_doc.add_settings_to_command("deposit", self) self.log.info(f"## Load deposit plugin {plugin_name}") # load plugin @@ -150,9 +371,43 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.info(f"## Run deposit plugin {plugin_name}") # run plugin try: - plugin_func(self) + plugin_func(self, prov_doc) except HermesValidationError as e: self.log.critical(f"## Error while executing {plugin_name} plugin.", exc_info=1) raise HermesPluginRunError( f"Something went wrong while running the deposit plugin {self.settings.plugin}" ) from e + + if prov_doc is None: + return + + # store provenance result + ctx = HermesCacheManager() + ctx.prepare_step("deposit") + with ctx["provenance"] as cache: + cache["result"] = prov_doc.ld_value + ctx.finalize_step("deposit") + + def load_prov_doc(self: Self) -> Optional[ld_prov_list]: + """ + Loads the provenance document of the curate step. + + Returns: + ld_prov_list | None: The loaded provenance document or None if the load failed. + """ + # set up HermesCache + ctx = HermesCacheManager() + ctx.prepare_step("curate") + with ctx["provenance"] as cache: + # try load + try: + return ld_prov_list.load_ld_prov_list(cache["result"]) + except Exception: + # log the warning and return None + self.log.warning( + "The provenance data from the curate step could not be loaded. " + "Deposition will proceed without collecting provenance data.", + exc_info=1 + ) + finally: + ctx.finalize_step("curate") diff --git a/src/hermes/commands/deposit/invenio.py b/src/hermes/commands/deposit/invenio.py index c1fa5870..3eaefd84 100644 --- a/src/hermes/commands/deposit/invenio.py +++ b/src/hermes/commands/deposit/invenio.py @@ -266,39 +266,6 @@ def __init__(self) -> None: self.invenio_ctx = None - def __call__(self, command, *, client=None, resolver=None): - self.command = command - self.config = getattr(self.command.settings, self.platform_name) - - if client is None: - auth_token = self.config.auth_token - - # TODO reactivate this code again, once we use Zenodo OAuth again (once the refresh token works) - # If auth_token is a refresh-token, get the auth-token from that. - # if str(auth_token).startswith("REFRESH_TOKEN:"): - # _log.debug(f"Getting token from refresh_token {auth_token}") - # # TODO How do we know if this targets sandbox or not? - # # Now we assume it's sandbox - # connect_zenodo.setup(True) - # tokens = connect_zenodo.oauth_process() \ - # .get_tokens_from_refresh_token(auth_token.split("REFRESH_TOKEN:")[1]) - # _log.debug(f"Tokens: {str(tokens)}") - # auth_token = tokens.get("access_token", "") - # _log.debug(f"Auth Token: {auth_token}") - # # TODO Update the secret (github/lab token is needed) - - if not auth_token: - raise DepositionUnauthorizedError("No valid auth token given for deposition platform") - self.client = self.invenio_client_class(self.config, - auth_token=auth_token, platform_name=self.platform_name) - else: - self.client = client - - self.resolver = resolver or self.invenio_resolver_class(self.client) - self.links = {} - - super().__call__(command) - # TODO: Populate some data structure here? Or move more of this into __init__.py? def prepare(self) -> None: """Prepare the deposition on an Invenio-based platform. @@ -314,6 +281,30 @@ def prepare(self) -> None: - check whether required configuration options are present - update ``self.metadata`` with metadata collected during the checks """ + self.config = getattr(self.command.settings, self.platform_name) + + auth_token = self.config.auth_token + + # TODO reactivate this code again, once we use Zenodo OAuth again (once the refresh token works) + # If auth_token is a refresh-token, get the auth-token from that. + # if str(auth_token).startswith("REFRESH_TOKEN:"): + # _log.debug(f"Getting token from refresh_token {auth_token}") + # # TODO How do we know if this targets sandbox or not? + # # Now we assume it's sandbox + # connect_zenodo.setup(True) + # tokens = connect_zenodo.oauth_process() \ + # .get_tokens_from_refresh_token(auth_token.split("REFRESH_TOKEN:")[1]) + # _log.debug(f"Tokens: {str(tokens)}") + # auth_token = tokens.get("access_token", "") + # _log.debug(f"Auth Token: {auth_token}") + # # TODO Update the secret (github/lab token is needed) + + if not auth_token: + raise DepositionUnauthorizedError("No valid auth token given for deposition platform") + self.client = self.invenio_client_class(self.config, auth_token=auth_token, platform_name=self.platform_name) + + self.resolver = self.invenio_resolver_class(self.client) + self.links = {} conf_rec_id = self.config.record_id conf_doi = self.config.doi diff --git a/src/hermes/commands/harvest/base.py b/src/hermes/commands/harvest/base.py index fec5746c..5bf92648 100644 --- a/src/hermes/commands/harvest/base.py +++ b/src/hermes/commands/harvest/base.py @@ -3,8 +3,14 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileContributor: Michael Meinel +# SPDX-FileContributor: Michael Fritzsche import argparse +import datetime +from io import IOBase +from pathlib import Path +from typing import Any, Callable, Optional +from typing_extensions import Self from pydantic import BaseModel @@ -12,34 +18,174 @@ from hermes.error import HermesPluginRunError, MisconfigurationError from hermes.model.hermes_cache import HermesCacheManager from hermes.model import SoftwareMetadata +from hermes.model.provenance.ld_prov import ld_prov_list class HermesHarvestPlugin(HermesPlugin): """Base plugin that does harvesting. + Attributes: + operations (list[tuple[dict[str, str], dict[str, str], dict[str, str]]]): The information recorded on the + load operations executed by the plugin. + TODO: describe the harvesting process and how this is mapped to this plugin. """ - def __call__(self, command: HermesCommand) -> SoftwareMetadata: + def __init__(self: Self) -> None: + """ + Create a new instance of a HermesHarvestPlugin. + + Returns: + None: + """ + self.operations: list[tuple[dict[str, str], dict[str, str], dict[str, str]]] = [] + super().__init__() + + def __call__(self: Self, command: "HermesHarvestCommand") -> SoftwareMetadata: + """ + Execute the hermes harvest plugin `self`. + + Args: + command (HermesHarvestCommand): The command being executed. + + Returns: + SoftwareMetadata: The harvested metadata. + """ pass + def load(self: Self, func: Callable, source: Any, *args: Optional[Any], **kwargs: Optional[Any]) -> Any: + """ + Load some data from some source using some function so that the calls provenance information is recorded. + + `func(source, *args, **kwargs)` will be executed. + + Args: + func (Callable): The function used for loading the requested source. + source (Any): The source the data is to be loaded from. + args (Any | None): Additional positional arguments for the load. + kwargs (Any | None): Additional keyword arguments for the load. + + Returns: + Any: The result of the load operation. + """ + # collect basic metadata + source_metadata = {"schema:description": "metadata source"} + if isinstance(source, IOBase): + source_metadata["schema:url"] = Path(source.name).absolute().as_uri() + elif isinstance(source, Path): + source_metadata["schema:url"] = source.absolute().as_uri() + elif isinstance(source, str): + try: + source_metadata["schema:url"] = Path(source).absolute().as_uri() + except Exception: + source_metadata["schema:url"] = source + operation = { + "schema:description": "Load operation called with (" + f"{source_metadata['schema:url'] if 'schema:url' in source_metadata else str(source)}" + f"{', ' + str(args) if args else ''}{', ' + str(kwargs) if kwargs else ''}).", + "schema:name": f"{func.__module__}.{func.__qualname__}" + } + operation["prov:startedAtTime"] = datetime.datetime.now() + # execute the load operation + result = func(source, *args, **kwargs) + # complete metadata collection + operation["prov:endedAtTime"] = datetime.datetime.now() + loaded_metadata = {"schema:description": "the loaded data", "schema:text": str(result)} + # store metadata + self.operations.append((source_metadata, operation, loaded_metadata)) + # return result of the load operation + return result + class HarvestSettings(BaseModel): - """Generic harvesting settings.""" + """ + Generic harvesting settings. + + Attributes: + sources (list[str]): (class attribute) A list of plugins to be executed. + """ sources: list[str] = [] +def remove_harvest_plugin_from_prov_doc(prov_doc: ld_prov_list, plugin: str) -> None: + """ + Removes information on the specified harvest plugin from the given provenance document. + + Args: + prov_doc (ld_prov_list): The provenance document the plugins information is to be removed from. + plugin (str): The name of the plugin of which the information is to be removed. + + Returns: + None: + """ + # get the plugin object from the prov_doc + plugin = prov_doc.get_hermes_plugin("harvest", plugin) + # If the plugin isn't contained in the prov_doc, return, otherwise fetch related objects + if plugin is None: + return + related = prov_doc.shallow_search(lambda node: ( + ("prov:wasAssociatedWith" in node and plugin.ref in node["prov:wasAssociatedWith"]) or + ("prov:wasAttributedTo" in node and plugin.ref in node["prov:wasAttributedTo"]) + )) + # If no related objects exist, delete only the plugin + if len(related) == 0: + del prov_doc[plugin.index] + return + # Collect remaining related objects + ids = [plugin.ref, *(rel.ref for rel in related)] + used_entities = [rel["prov:used"][0]["@id"] for rel in related if "prov:used" in rel] + related = prov_doc.shallow_search(lambda node: node["@id"] in used_entities) + related += prov_doc.shallow_search(lambda node: any( + (f"prov:{key}" in node and id in node[f"prov:{key}"]) for id in ids for key in [ + "wasAssociatedWith", "wasAttributedTo", "wasGeneratedBy", "used", "wasDerivedFrom", "wasInformedBy" + ] + )) + # delete all collected objects + del prov_doc[plugin.index] + for item in related: + items = prov_doc.shallow_search(lambda node: ("@id" in node and node["@id"] == item["@id"])) + if len(items) == 1: + del prov_doc[items[0].index] + + class HermesHarvestCommand(HermesCommand): - """ Harvest metadata from configured sources. """ + """ + Harvest metadata from configured sources. - command_name = "harvest" - settings_class = HarvestSettings + Attributes: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + command_name (str): (class attribute) The name of the command + settings_class (type): (class attribute) The settings class for general harvest settings. + """ - def __call__(self, args: argparse.Namespace) -> None: - self.log.info("# Metadata harvesting") + command_name: str = "harvest" + settings_class: type = HarvestSettings + + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + + Raises: + MisconfigurationError: If no plugin is configured to be run. + HermesPluginRunError: If all plugin runs failed. + """ self.args = args + self.log.info("# Load provenance from old harvest or create new document.") + # initialize the provenance document for this run + prov_doc = self.init_provenance_document() + prov_doc.add_hermes_settings(self) + prov_doc.add_settings_to_command("harvest", self) + # get basic hermes object to reference later + base_plugin = prov_doc.get_hermes_base_plugin("harvest") + self.log.info("# Metadata harvesting") if len(self.settings.sources) == 0: self.log.critical("# No harvest plugin was configured to be run and loaded.") raise MisconfigurationError("No harvest plugin was configured to be run and loaded.") @@ -54,7 +200,7 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.info(f"### Load {plugin_name} plugin") # load plugin try: - plugin_func = self.plugins[plugin_name]() + plugin_func: HermesHarvestPlugin = self.plugins[plugin_name]() except KeyError: self.log.error(f"### Plugin {plugin_name} not found, skipping it now.") continue @@ -62,17 +208,136 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.info(f"### Run {plugin_name} plugin") # run plugin try: - harvested_data = plugin_func(self) + harvested_data: SoftwareMetadata = plugin_func(self) except Exception: self.log.exception(f"### Unknown error while executing the {plugin_name} plugin, skipping it now.") continue + returned_at_time = datetime.datetime.now() self.log.info(f"### Store metadata harvested by {plugin_name} plugin") # store harvested data + begin_store_at_time = datetime.datetime.now() harvested_data.write_to_cache(ctx, plugin_name) + stored_at_time = datetime.datetime.now() harvested_any = True + # remove old provenance data from a potential existent old run of this plugin + remove_harvest_plugin_from_prov_doc(prov_doc, plugin_name) + + # add the plugins provenance information + plugin = prov_doc.add_hermes_plugin("harvest", plugin_name, plugin_func, self) + # add the collected information on the load operations of the plugin to the provenance document + plugin_operations = plugin_func.operations + outputs, io_ops = [], [] + for plugin_operation in plugin_operations: + loaded_source = prov_doc.add_entity(data=plugin_operation[0]) + plugin_operation[1].update( + {"prov:wasAssociatedWith": [base_plugin.ref, plugin.ref], "prov:used": loaded_source.ref} + ) + io_op = prov_doc.add_activity(data=plugin_operation[1]) + plugin_operation[2].update({ + "prov:wasAttributedTo": plugin.ref, + "prov:wasDerivedFrom": loaded_source.ref, + "prov:wasGeneratedBy": io_op.ref + }) + loaded_data = prov_doc.add_entity(data=plugin_operation[2]) + # store references to the added objects + outputs.append(loaded_data.ref) + io_ops.append(io_op.ref) + + # add provenance information on the mapping and returned data + map_activity = prov_doc.add_activity(data={ + "schema:description": "Maps the loaded data to the JSON-LD contexts vocabulary.", + "prov:wasInformedBy": io_ops, + "prov:used": outputs, + "prov:wasAssociatedWith": plugin.ref, + "prov:endedAtTime": returned_at_time + }) + data_output = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "the harvested metadata", + "schema:text": str(harvested_data.compact()), + "prov:wasAttributedTo": plugin.ref, + "prov:wasGeneratedBy": map_activity.ref, + "prov:wasDerivedFrom": outputs, + "prov:generatedAtTime": returned_at_time + }) + + # add provenance information on the write and stored data + write = prov_doc.add_activity(data={ + "schema:description": "Writes the harvested metadata into the HERMES cache.", + "prov:wasAssociatedWith": [ + prov_doc.get_hermes_command("harvest").ref, + prov_doc.get_hermes_cache().ref, + plugin.ref + ], + "prov:used": data_output.ref, + "prov:wasInformedBy": map_activity.ref, + "prov:startedAtTime": begin_store_at_time, + "prov:endedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The compacted version of the harvested metadata.", + "schema:text": str(harvested_data.compact()), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "harvest" / plugin_name / "codemeta.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": data_output.ref, + "prov:wasAttributedTo": prov_doc.get_hermes_cache().ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The context of the harvested metadata.", + "schema:text": str({"@context": harvested_data.full_context}), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "harvest" / plugin_name / "context.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": data_output.ref, + "prov:wasAttributedTo": prov_doc.get_hermes_cache().ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The expanded version of the harvested metadata.", + "schema:text": str(harvested_data.ld_value), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "harvest" / plugin_name / "expanded.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": data_output.ref, + "prov:wasAttributedTo": prov_doc.get_hermes_cache().ref, + "prov:generatedAtTime": stored_at_time + }) + + # store provenance information + with ctx["provenance"] as cache: + cache["result"] = prov_doc.ld_value + ctx.finalize_step('harvest') if not harvested_any: self.log.critical("No harvest plugin ran successfully.") raise HermesPluginRunError("No harvest plugin ran successfully.") + + @classmethod + def init_provenance_document(cls: type[Self]) -> ld_prov_list: + """ + Loads or creates a provenance document. + + Returns: + ld_prov_list: The loaded or created provenance document. + """ + # try loading the document + ctx = HermesCacheManager() + ctx.prepare_step("harvest") + with ctx["provenance"] as cache: + try: + return ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + pass + finally: + ctx.finalize_step("harvest") + # initialize a new ld_prov_list because load failed + prov_doc = ld_prov_list() + prov_doc.init_hermes_agents() + return prov_doc diff --git a/src/hermes/commands/harvest/cff.py b/src/hermes/commands/harvest/cff.py index 5a2d16c1..f2b648e6 100644 --- a/src/hermes/commands/harvest/cff.py +++ b/src/hermes/commands/harvest/cff.py @@ -43,7 +43,7 @@ def __call__(self, command: HermesHarvestCommand) -> tuple[SoftwareMetadata, dic 'Aborting harvesting for this metadata source.') # Read the content - cff_data = cff_file.read_text() + cff_data = self.load(pathlib.Path.read_text, cff_file) cff_dict = self._load_cff_from_file(cff_data) if command.settings.cff.enable_validation: diff --git a/src/hermes/commands/harvest/codemeta.py b/src/hermes/commands/harvest/codemeta.py index 3dc84296..07645647 100644 --- a/src/hermes/commands/harvest/codemeta.py +++ b/src/hermes/commands/harvest/codemeta.py @@ -34,7 +34,7 @@ def __call__(self, command: HermesHarvestCommand) -> tuple[SoftwareMetadata, dic ) # Read the content - codemeta_str = codemeta_file.read_text() + codemeta_str = self.load(pathlib.Path.read_text, codemeta_file) if not self._validate(codemeta_file): raise HermesValidationError(codemeta_file) diff --git a/src/hermes/commands/postprocess/base.py b/src/hermes/commands/postprocess/base.py index 99a26d73..1ab826db 100644 --- a/src/hermes/commands/postprocess/base.py +++ b/src/hermes/commands/postprocess/base.py @@ -6,36 +6,225 @@ # SPDX-FileContributor: Michael Fritzsche import argparse +import datetime +from io import IOBase +from pathlib import Path +from typing import Any, Callable, Optional +from typing_extensions import Self from pydantic import BaseModel from hermes.commands.base import HermesCommand, HermesPlugin from hermes.error import HermesPluginRunError +from hermes.model.hermes_cache import HermesCacheManager +from hermes.model.provenance.ld_prov import ld_prov_list +from hermes.model.types import ld_dict class HermesPostprocessPlugin(HermesPlugin): - """ Base plugin for postprocess plugins. """ + """ + Base plugin for postprocess plugins. - def __call__(self, command: HermesCommand) -> None: + Attributes: + cache_operations (list[tuple[str, dict[str, str], dict[str, str]]]): The information recorded on the + cache load operations executed by the plugin. + load_operations (list[tuple[dict[str, str], dict[str, str], dict[str, str]]]): The information recorded on the + load operations executed by the plugin. + write_operations (list[tuple[dict[str, str], dict[str, str], dict[str, str]]]): The information recorded on the + write operations executed by the plugin. + """ + + def __init__(self: Self) -> None: + """ + Create a new instance of a HermesPostprocessPlugin. + + Returns: + None: + """ + self.cache_operations: list[tuple[str, dict[str, str], dict[str, str]]] = [] + self.load_operations: list[tuple[dict[str, str], dict[str, str], dict[str, str]]] = [] + self.write_operations: list[tuple[dict[str, str], dict[str, str], dict[str, str]]] = [] + super().__init__() + + def __call__(self: Self, command: "HermesPostprocessCommand") -> None: + """ + Execute the hermes postprocess plugin `self`. + + Args: + command (HermesPostprocessCommand): The command being executed. + + Returns: + None: + """ pass + def get_deposit_result(self: Self, target: str) -> dict: + """ + Load the result of some deposit plugin from the cache so that the calls provenance information is recorded. + + Args: + target (str): The name of the deposit plugin. + + Returns: + dict: The result of the cache load operation. + """ + # collect basic metadata + source_metadata = target[:] + load_operation = {"schema:description": f"loads the result of deposit plugin {target}"} + ctx = HermesCacheManager() + ctx.prepare_step("deposit") + load_operation["prov:startedAtTime"] = datetime.datetime.now() + # execute the load operation + with ctx[target] as cache: + res = cache["result"] + # complete metadata collection + load_operation["prov:endedAtTime"] = datetime.datetime.now() + ctx.finalize_step("deposit") + loaded_data = {"schema:description": "the loaded data", "schema:text": str(res)} + # store metadata + self.cache_operations.append((source_metadata, load_operation, loaded_data)) + # return result of the load operation + return res + + def load(self: Self, func: Callable, source: Any, *args: Optional[Any], **kwargs: Optional[Any]) -> Any: + """ + Load some data from some source using some function so that the calls provenance information is recorded. + + `func(source, *args, **kwargs)` will be executed. + + Args: + func (Callable): The function used for loading the requested source. + source (Any): The source the data is to be loaded from. + args (Any | None): Additional positional arguments for the load. + kwargs (Any | None): Additional keyword arguments for the load. + + Returns: + Any: The result of the load operation. + """ + # collect basic metadata + source_metadata = {"schema:description": "metadata source"} + if isinstance(source, IOBase): + source_metadata["schema:url"] = Path(source.name).absolute().as_uri() + elif isinstance(source, Path): + source_metadata["schema:url"] = source.absolute().as_uri() + elif isinstance(source, str): + try: + source_metadata["schema:url"] = Path(source).absolute().as_uri() + except Exception: + source_metadata["schema:url"] = source + load_operation = { + "schema:description": "Load operation called with (" + f"{source_metadata['schema:url'] if 'schema:url' in source_metadata else str(source)}" + f"{', ' + str(args) if args else ''}{', ' + str(kwargs) if kwargs else ''}).", + "schema:name": f"{func.__module__}.{func.__qualname__}" + } + load_operation["prov:startedAtTime"] = datetime.datetime.now() + # execute the load operation + result = func(source, *args, **kwargs) + # complete metadata collection + load_operation["prov:endedAtTime"] = datetime.datetime.now() + loaded_metadata = {"schema:description": "the loaded data", "schema:text": str(result)} + # store metadata + self.load_operations.append((source_metadata, load_operation, loaded_metadata)) + # return result of the load operation + return result + + def write( + self: Self, func: Callable, data: Any, destination: Any, *args: Optional[Any], **kwargs: Optional[Any] + ) -> Any: + """ + Write some data from some source using some function so that the calls provenance information is recorded. + + `func(source, *args, **kwargs)` will be executed. + + Args: + func (Callable): The function used for writing the requested source. + data (Any): The data that is to be written. + destination (Any): The source the data is to be written to. + args (Any | None): Additional positional arguments for the write. + kwargs (Any | None): Additional keyword arguments for the write. + + Returns: + Any: The result of the write operation. + """ + # collect basic metadata + destination_metadata = {"schema:description": "metadata destination"} + if isinstance(destination, IOBase): + destination_metadata["schema:url"] = Path(destination.name).absolute().as_uri() + elif isinstance(destination, Path): + destination_metadata["schema:url"] = destination.absolute().as_uri() + elif isinstance(destination, str): + try: + destination_metadata["schema:url"] = Path(destination).absolute().as_uri() + except Exception: + destination_metadata["schema:url"] = destination + write_operation = { + "schema:description": f"Write operation called with ({str(data)}," + f"{destination_metadata['schema:url'] if 'schema:url' in destination_metadata else str(destination)}" + f"{', ' + str(args) if args else ''}{', ' + str(kwargs) if kwargs else ''}).", + "schema:name": f"{func.__module__}.{func.__qualname__}" + } + write_operation["prov:startedAtTime"] = datetime.datetime.now() + # execute the write operation + result = func(data, destination, *args, **kwargs) + # complete metadata collection + write_operation["prov:endedAtTime"] = datetime.datetime.now() + written_metadata = {"schema:description": "the written data", "schema:text": str(data)} + # store metadata + self.write_operations.append((written_metadata, write_operation, destination_metadata)) + # return result of the write operation + return result + class PostprocessSettings(BaseModel): - """Generic post-processing settings.""" + """ + Generic post-processing settings. + + Attributes: + run (list[str]): A list of plugins to be executed. + """ - run: list = [] + run: list[str] = [] class HermesPostprocessCommand(HermesCommand): - """Post-process the published metadata after deposition.""" + """ + Post-process the published metadata after deposition. + + Attributes: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general deposit settings. + """ + + command_name: str = "postprocess" + settings_class: type = PostprocessSettings + + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. - command_name = "postprocess" - settings_class = PostprocessSettings + Returns: + None: - def __call__(self, args: argparse.Namespace) -> None: + Raises: + HermesPluginRunError: If something went wrong with all plugin runs. + """ self.log.info("# Postprocessing") self.args = args plugin_names = self.settings.run + # try loading and adding general information to the provenance document + prov_doc = self.load_prov_doc() + if prov_doc is not None: + prov_doc.add_hermes_settings(self) + prov_doc.add_settings_to_command("postprocess", self) + # get basic hermes objects to reference later + hermes_cache = prov_doc.get_hermes_cache() + postprocess_command = prov_doc.get_hermes_command("postprocess") + postprocess_base_plugin = prov_doc.get_hermes_base_plugin("postprocess") if not plugin_names: self.log.warning("# No plugin was configured to be run yet the postprocess command was executed.") @@ -47,7 +236,7 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.info(f"### Load {plugin_name} plugin") # load plugin try: - plugin_func = self.plugins[plugin_name]() + plugin_func: HermesPostprocessPlugin = self.plugins[plugin_name]() except KeyError: self.log.error(f"### Plugin {plugin_name} not found.") continue @@ -62,6 +251,103 @@ def __call__(self, args: argparse.Namespace) -> None: ran_any = True + if prov_doc is None: + continue + + # add information on the postprocess plugin + plugin = prov_doc.add_hermes_plugin("postprocess", plugin_name, plugin_func, self) + # add the collected information on the io operations of the plugin to the provenance document + cache_loads = plugin_func.cache_operations + loads = plugin_func.load_operations + writes = plugin_func.write_operations + load_actions: list[ld_dict] = [] + loaded_datas: list[ld_dict] = [] + # add cache load operations to the provenance document + for cache_load in cache_loads: + deposit_plugin = prov_doc.get_hermes_plugin("deposit", cache_load[0]) + updated_metadata = prov_doc.shallow_search(lambda node: ( + "prov:wasInfluencedBy" in node and node["prov:wasInfluencedBy"] == [deposit_plugin.ref] + ))[0] + updated_metadata = prov_doc.shallow_search(lambda node: ( + "prov:wasDerivedFrom" in node and node["prov:wasDerivedFrom"] == [updated_metadata.ref] + ))[0] + load_actions.append(prov_doc.add_activity(data=cache_load[1])) + load_actions[-1].update({ + "prov:used": updated_metadata.ref, + "prov:wasAssociatedWith": [ + plugin.ref, postprocess_base_plugin.ref, postprocess_command.ref, hermes_cache.ref + ] + }) + loaded_datas.append(prov_doc.add_entity(data=cache_load[2])) + loaded_datas[-1].update({ + "prov:wasGeneratedBy": load_actions[-1].ref, + "prov:wasDerivedFrom": updated_metadata.ref, + "prov:wasAttributedTo": hermes_cache.ref + }) + # add load operations to the provenance document + for load in loads: + source = prov_doc.add_entity(data=load[0]) + load_actions.append(prov_doc.add_activity(data=load[1])) + load_actions[-1].update({ + "prov:used": source.ref, + "prov:wasAssociatedWith": [plugin.ref, postprocess_base_plugin.ref, postprocess_command.ref] + }) + loaded_datas.append(prov_doc.add_entity(data=load[2])) + loaded_datas[-1].update({ + "prov:wasGeneratedBy": load_actions[-1].ref, + "prov:wasDerivedFrom": source.ref, + "prov:wasAttributedTo": [plugin.ref, postprocess_base_plugin.ref, postprocess_command.ref] + }) + load_actions = [load_action.ref for load_action in load_actions] + loaded_datas = [loaded_data.ref for loaded_data in loaded_datas] + # add write operations to the provenance document + for write in writes: + data = prov_doc.add_entity(data=write[0]) + data.update({"prov:wasDerivedFrom": loaded_datas, "prov:wasInfluencedBy": plugin.ref}) + write_action = prov_doc.add_activity(data=write[1]) + write_action.update({ + "prov:used": data.ref, + "prov:wasAssociatedWith": [plugin.ref, postprocess_base_plugin.ref, postprocess_command.ref] + }) + prov_doc.add_entity(data=write[2]).update({ + "prov:wasGeneratedBy": write_action.ref, + "prov:wasDerivedFrom": data.ref, + "prov:wasAttributedTo": [plugin.ref, postprocess_base_plugin.ref, postprocess_command.ref] + }) + + if prov_doc is not None: + # store provenance data + ctx = HermesCacheManager() + ctx.prepare_step("postprocess") + with ctx["provenance"] as cache: + cache["result"] = prov_doc.ld_value + ctx.finalize_step("postprocess") + + # error out if no plugin ran successfully if not ran_any: self.log.critical("## No postprocess plugin ran successfully.") raise HermesPluginRunError("No postprocess plugin ran successfully.") + + def load_prov_doc(self: Self) -> Optional[ld_prov_list]: + """ + Loads the provenance document of the postprocess step. + + Returns: + ld_prov_list | None: The loaded provenance document or None if the load failed. + """ + # set up HermesContext + ctx = HermesCacheManager() + ctx.prepare_step("deposit") + with ctx["provenance"] as cache: + # try load + try: + return ld_prov_list.load_ld_prov_list(cache["result"]) + except Exception: + # log the warning and return None + self.log.warning( + "The provenance data from the deposit step could not be loaded. " + "Postprocessing will proceed without collecting provenance data.", + exc_info=1 + ) + finally: + ctx.finalize_step("deposit") diff --git a/src/hermes/commands/postprocess/invenio.py b/src/hermes/commands/postprocess/invenio.py index 13e952a5..e2baa80c 100644 --- a/src/hermes/commands/postprocess/invenio.py +++ b/src/hermes/commands/postprocess/invenio.py @@ -13,7 +13,7 @@ import tomlkit from hermes.error import MisconfigurationError -from hermes.model.hermes_cache import HermesCacheManager + from ..base import HermesCommand from .base import HermesPostprocessPlugin @@ -23,13 +23,9 @@ class config_record_id(HermesPostprocessPlugin): def __call__(self, command: HermesCommand): - ctx = HermesCacheManager() - ctx.prepare_step("deposit") - with ctx["invenio"] as manager: - deposition = manager["result"] - ctx.finalize_step("deposit") + deposition = self.get_deposit_result("invenio") - conf = tomlkit.load(open('hermes.toml', 'r')) + conf = self.load(tomlkit.load, open('hermes.toml', 'r')) try: old_record_id = conf["deposit"]["invenio"]["record_id"] if old_record_id == deposition["record_id"]: @@ -42,16 +38,13 @@ def __call__(self, command: HermesCommand): except KeyError: pass conf.setdefault("deposit", {}).setdefault("invenio", {})["record_id"] = deposition['record_id'] - tomlkit.dump(conf, open('hermes.toml', 'w')) + self.write(tomlkit.dump, conf, open('hermes.toml', 'w')) class cff_doi(HermesPostprocessPlugin): def __call__(self, command: HermesCommand): - ctx = HermesCacheManager() - ctx.prepare_step("deposit") - with ctx["invenio"] as manager: - deposition = manager["result"] - ctx.finalize_step("deposit") + + deposition = self.get_deposit_result("invenio") yaml = YAML() yaml.default_flow_style = False @@ -60,7 +53,7 @@ def __call__(self, command: HermesCommand): yaml.allow_unicode = True try: - cff = yaml.load(open('CITATION.cff', 'r')) + cff = self.load(yaml.load, open('CITATION.cff', 'r')) new_identifier = { 'description': f"DOI for the published version {deposition['metadata']['version']} " "[generated by hermes]", @@ -71,22 +64,18 @@ def __call__(self, command: HermesCommand): cff['identifiers'].append(new_identifier) else: cff['identifiers'] = [new_identifier] - yaml.dump(cff, open('CITATION.cff', 'w')) + self.write(yaml.dump, cff, open('CITATION.cff', 'w')) except Exception as e: raise RuntimeError("Update of CITATION.cff failed.") from e class codemeta_doi(HermesPostprocessPlugin): def __call__(self, command: HermesCommand): - ctx = HermesCacheManager() - ctx.prepare_step("deposit") - with ctx["invenio"] as manager: - deposition = manager["result"] - ctx.finalize_step("deposit") + deposition = self.get_deposit_result("invenio") try: with open("codemeta.json", "r") as file: - codemeta = json.load(file) + codemeta = self.load(json.load, file) if "@id" not in codemeta: codemeta["@id"] = deposition['doi'] if "referencePublication" not in codemeta: @@ -96,6 +85,6 @@ def __call__(self, command: HermesCommand): else: codemeta["referencePublication"] = [codemeta["referencePublication"], deposition['doi']] with open("codemeta.json", "w") as file: - json.dump(codemeta, file) + self.write(json.dump, codemeta, file) except Exception as e: raise RuntimeError("Update of CITATION.cff failed.") from e diff --git a/src/hermes/commands/postprocess/invenio_rdm.py b/src/hermes/commands/postprocess/invenio_rdm.py index f882d763..ce3c6b73 100644 --- a/src/hermes/commands/postprocess/invenio_rdm.py +++ b/src/hermes/commands/postprocess/invenio_rdm.py @@ -6,12 +6,13 @@ # SPDX-FileContributor: Michael Fritzsche # SPDX-FileContributor: Stephan Druskat +import json import logging import tomlkit +from ruamel.yaml import YAML from hermes.error import MisconfigurationError -from hermes.model.hermes_cache import HermesCacheManager from ..base import HermesCommand from .base import HermesPostprocessPlugin @@ -21,13 +22,9 @@ class config_record_id(HermesPostprocessPlugin): def __call__(self, command: HermesCommand): - ctx = HermesCacheManager() - ctx.prepare_step("deposit") - with ctx["invenio_rdm"] as manager: - deposition = manager["result"] - ctx.finalize_step("deposit") + deposition = self.get_deposit_result("invenio_rdm") - conf = tomlkit.load(open('hermes.toml', 'r')) + conf = self.load(tomlkit.load, open('hermes.toml', 'r')) try: old_record_id = conf["deposit"]["invenio_rdm"]["record_id"] if old_record_id == deposition["record_id"]: @@ -40,4 +37,53 @@ def __call__(self, command: HermesCommand): except KeyError: pass conf.setdefault("deposit", {}).setdefault("invenio_rdm", {})["record_id"] = deposition['record_id'] - tomlkit.dump(conf, open('hermes.toml', 'w')) + self.write(tomlkit.dump, conf, open('hermes.toml', 'w')) + + +class cff_doi(HermesPostprocessPlugin): + def __call__(self, command: HermesCommand): + + deposition = self.get_deposit_result("invenio_rdm") + + yaml = YAML() + yaml.default_flow_style = False + yaml.allow_unicode = True + yaml.indent(mapping=4, sequence=2, offset=0) + yaml.allow_unicode = True + + try: + cff = self.load(yaml.load, open('CITATION.cff', 'r')) + new_identifier = { + 'description': f"DOI for the published version {deposition['metadata']['version']} " + "[generated by hermes]", + 'type': 'doi', + 'value': deposition['metadata']['prereserve_doi']['doi'] + } + if 'identifiers' in cff: + cff['identifiers'].append(new_identifier) + else: + cff['identifiers'] = [new_identifier] + self.write(yaml.dump, cff, open('CITATION.cff', 'w')) + except Exception as e: + raise RuntimeError("Update of CITATION.cff failed.") from e + + +class codemeta_doi(HermesPostprocessPlugin): + def __call__(self, command: HermesCommand): + deposition = self.get_deposit_result("invenio_rdm") + doi = deposition['metadata']['prereserve_doi']['doi'] + try: + with open("codemeta.json", "r") as file: + codemeta = self.load(json.load, file) + if "@id" not in codemeta: + codemeta["@id"] = doi + if "referencePublication" not in codemeta: + codemeta["referencePublication"] = doi + elif isinstance(codemeta["referencePublication"], list): + codemeta["referencePublication"].append(doi) + else: + codemeta["referencePublication"] = [codemeta["referencePublication"], doi] + with open("codemeta.json", "w") as file: + self.write(json.dump, codemeta, file) + except Exception as e: + raise RuntimeError("Update of CITATION.cff failed.") from e diff --git a/src/hermes/commands/process/base.py b/src/hermes/commands/process/base.py index 884b3278..2c7f21ef 100644 --- a/src/hermes/commands/process/base.py +++ b/src/hermes/commands/process/base.py @@ -5,16 +5,22 @@ # SPDX-FileContributor: Michael Meinel import argparse +import datetime +from typing import Optional +from typing_extensions import Self from typing import TypeAlias, Union from pydantic import BaseModel from hermes.commands.base import HermesCommand, HermesPlugin +from hermes.commands.harvest.base import remove_harvest_plugin_from_prov_doc from hermes.error import HermesPluginRunError, MisconfigurationError from hermes.model.api import SoftwareMetadata from hermes.model.hermes_cache import HermesCacheManager from hermes.model.merge.action import MergeAction from hermes.model.merge.container import ld_merge_dict +from hermes.model.provenance.ld_prov import ld_prov_list +from hermes.model.types import ld_dict TypeIRI: TypeAlias = Union[str, None] """ Type description for the iri of a JSON-LD property or object type (or None indicating a 'joker') """ @@ -25,28 +31,75 @@ class HermesProcessPlugin(HermesPlugin): - """ Base plugin that defines additional merge strategies.""" + """ Base plugin that defines additional merge strategies. """ + + def __call__(self: Self, command: "HermesProcessCommand") -> dict[Optional[str], dict[Optional[str], MergeAction]]: + """ + Execute the hermes process plugin `self`. + + Args: + command (HermesProcessCommand): The command being executed. + + Returns: + dict[str | None, dict[str | None, MergeAction]]: The merge strategies. + """ - def __call__(self, command: HermesCommand) -> ObjectStrategies: pass class ProcessSettings(BaseModel): - """Generic deposition settings.""" + """ + Generic deposition settings. - sources: list = [] - plugins: list = ["codemeta"] + Attributes: + sources (list[str]): (class attribute) A list of harvest plugins whoose results should be processed. + plugins (list[str]): (class attribute) A list of plugins to be executed. + """ + + sources: list[str] = [] + plugins: list[str] = ["codemeta"] class HermesProcessCommand(HermesCommand): - """ Process the collected metadata into a common dataset. """ + """ + Process the collected metadata into a common dataset. + + Attributes: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + command_name (str): (class attribute) The name of the command + settings_class (type): (class attribute) The settings class for general process settings. + """ + + command_name: str = "process" + settings_class: type = ProcessSettings - command_name = "process" - settings_class = ProcessSettings + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + + Raises: + MisconfigurationError: If it was explicitly configured that no process plugin should be run. + MisconfigurationError: If no harvesters have been configured to be used. + """ + self.args = args + self.log.info("# Load provenance data from harvest step") + # try loading and adding general information to the provenance document + prov_doc = self.load_prov_doc() + if prov_doc is not None: + prov_doc.add_hermes_settings(self) + prov_doc.add_settings_to_command("process", self) + # get basic hermes objects to reference later + process_command = prov_doc.get_hermes_command("process") + hermes_cache = prov_doc.get_hermes_cache() - def __call__(self, args: argparse.Namespace) -> None: self.log.info("# Metadata processing") - merged_doc = ld_merge_dict([{}]) + merge_doc = ld_merge_dict([{}], prov_doc) if not self.settings.plugins: self.log.critical( @@ -55,14 +108,107 @@ def __call__(self, args: argparse.Namespace) -> None: ) raise MisconfigurationError("Explicit configuration to use no process plugin.") - # Get all harvesters + # Get all harvesters whoose results should be merged harvester_names = self.settings.sources if self.settings.sources else self.root_settings.harvest.sources if not harvester_names: self.log.critical("# No harvesters to merge from were configured.") raise MisconfigurationError("No harvesters to merge from were configured.") + # generate strategies and add them to the merge_doc + # add provenance information on the process to the prov_doc if provenance is recorded + strategy_action, merged_strategies = self.add_strategies_to_merge_doc(merge_doc, prov_doc) + + # load data and merge it + # add provenance information on the process to the prov_doc if provenance is recorded + last_action, last_data = self.merge_data_from_harvesters( + merge_doc, harvester_names, prov_doc, strategy_action, merged_strategies + ) + + # set up HermesContext + ctx = HermesCacheManager() + self.log.info("## Store processed metadata") + # store processed data + ctx.prepare_step("process") + begin_store_at_time = datetime.datetime.now() + with ctx["result"] as result_ctx: + result_ctx["codemeta"] = merge_doc.compact() + result_ctx["context"] = {"@context": merge_doc.full_context} + result_ctx["expanded"] = merge_doc.ld_value + stored_at_time = datetime.datetime.now() + + if prov_doc is not None: + # add provenance information on the write and stored data + write = prov_doc.add_activity(data={ + "schema:description": "Writes the processed metadata into the HERMES cache.", + "prov:wasAssociatedWith": [process_command.ref, hermes_cache.ref], + "prov:used": last_data.ref, + "prov:wasInformedBy": last_action.ref, + "prov:startedAtTime": begin_store_at_time, + "prov:endedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The compacted version of the processed metadata.", + "schema:text": str(merge_doc.compact()), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "process" / "result" / "codemeta.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": last_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The context of the processed metadata.", + "schema:text": str({"@context": merge_doc.full_context}), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "process" / "result" / "context.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": last_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": "The expanded version of the processed metadata.", + "schema:text": str(merge_doc.ld_value), + "schema:encodingFormat": "application/json", + "schema:url": (ctx.cache_dir / "process" / "result" / "expanded.json").absolute().as_uri(), + "prov:wasGeneratedBy": write.ref, + "prov:wasDerivedFrom": last_data.ref, + "prov:wasAttributedTo": hermes_cache.ref, + "prov:generatedAtTime": stored_at_time + }) + + # store provenance data + with ctx["provenance"] as cache: + cache["result"] = prov_doc.ld_value + + ctx.finalize_step("process") + + def add_strategies_to_merge_doc( + self: Self, merge_doc: ld_merge_dict, prov_doc: Optional[ld_prov_list] + ) -> tuple[Optional[ld_dict], Optional[ld_dict]]: + """ + Adds strategies to the merge doc that are generated by the process plugins and add provenance information to the + prov_doc. + + Args: + merge_doc (ld_merge_dict): The merge_doc the strategies are to be added to. + prov_doc (ld_prov_list | None): The provenance document where the information is to be recorded. + + Returns: + tuple[ld_dict | None, ld_dict | None]: The object of the last merge of strategies and the object of the + merged strategies. + + Raises: + HermesPluginRunError: If all plugin runs failed. + """ self.log.info("## Load and run the plugins") any_strategies_loaded = False + strategy_action, merged_strategies = None, None + if prov_doc is not None: + process_command = prov_doc.get_hermes_command("process") # add the strategies from the plugins for plugin_name in reversed(self.settings.plugins): self.log.info(f"### Load {plugin_name} plugin") @@ -76,45 +222,176 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.info(f"### Run {plugin_name} plugin") # run plugin try: + generate_strategies_start = datetime.datetime.now() additional_strategies = plugin_func(self) + generate_strategies_end = datetime.datetime.now() except Exception: self.log.exception(f"### Unknown error while executing the {plugin_name} plugin, skipping it now.") continue self.log.info(f"### Add the strategies to the merge document {plugin_name} plugin") # add strategies to the merge document - merged_doc.add_strategy(additional_strategies) + merge_strategies_start = datetime.datetime.now() + merge_doc.add_strategy(additional_strategies) + merge_strategies_end = datetime.datetime.now() any_strategies_loaded = True + if prov_doc is None: + continue + # add plugin and information on the generation of the merge strategies to the provenance document + plugin = prov_doc.add_hermes_plugin("process", plugin_name, plugin_func, self) + new_strategy_generation = prov_doc.add_activity(data={ + "schema:description": "generate new merge strategies", + "prov:wasAssociatedWith": plugin.ref, + "prov:startedAtTime": generate_strategies_start, + "prov:endedAtTime": generate_strategies_end + }) + new_strategies = prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": f"new merge strategies of plugin {plugin_name}", + "schema:text": str(additional_strategies), + "prov:wasAttributedTo": plugin.ref, + "prov:wasGeneratedBy": new_strategy_generation.ref, + "prov:generatedAtTime": generate_strategies_end + }) + if merged_strategies is None: + # only first pass for interiteration connections + merged_strategies = new_strategies + strategy_action = new_strategy_generation + continue + # add information on the merge of the old merge strategies with the new ones + strategy_action = prov_doc.add_activity(data={ + "schema:description": "merging the new strategies into the others", + "prov:used": [merged_strategies.ref, new_strategies.ref], + "prov:wasInformedBy": [strategy_action.ref, new_strategy_generation.ref], + "prov:wasAssociatedWith": process_command.ref, + "prov:startedAtTime": merge_strategies_start, + "prov:endedAtTime": merge_strategies_end + }) + merged_strategies = prov_doc.add_entity(data={ + "schema:description": "the merge strategies of multiple plugins merged together", + "schema:text": str(merge_doc.strategies), + "prov:wasDerivedFrom": [merged_strategies.ref, new_strategies.ref], + "prov:wasGeneratedBy": strategy_action.ref, + "prov:wasAttributedTo": process_command.ref, + "prov:generatedAtTime": merge_strategies_end + }) + + # error if no strategies could be loaded if not any_strategies_loaded: self.log.critical("## No process plugin was ran successfully.") raise HermesPluginRunError("No process plugin was ran successfully.") - ctx = HermesCacheManager() - ctx.prepare_step('harvest') + return strategy_action, merged_strategies + + def merge_data_from_harvesters( + self: Self, + merge_doc: ld_merge_dict, + harvester_names: list[str], + prov_doc: Optional[ld_prov_list], + strategy_action: Optional[ld_dict], + merged_strategies: Optional[ld_dict] + ) -> tuple[Optional[ld_dict], Optional[ld_dict]]: + """ + Load data from the harvest plugins and merge it and then add provenance information to the prov_doc. + + Args: + merge_doc (ld_merge_dict): The merge_doc the strategies are to be added to. + harvester_names (list[str]): The harvesters whoose data is to be loaded and merged. + prov_doc (ld_prov_list | None): The provenance document where the information is to be recorded. + strategy_action (ld_dict | None): The last action merging strategies (prov object). + merged_strategies (ld_dict | None): The result of the strategy merges (prov object). + + Returns: + tuple[ld_dict | None, ld_dict | None]: The object of the last merge and the object of the merged data. + + Raises: + RuntimeError: If a merge failed. + RuntimeError: If data from all harvesters couldn't be loaded. + """ + if prov_doc is not None: + process_command = prov_doc.get_hermes_command("process") + hermes_cache = prov_doc.get_hermes_cache() # merge data from harvesters self.log.info("## Merge the metadata of the harvesters") + ctx = HermesCacheManager() + ctx.prepare_step('harvest') merged_any = False + last_action, last_data = None, None for harvester in harvester_names: self.log.info(f"### Load data from {harvester} plugin") # load data from harvester try: + load_start = datetime.datetime.now() metadata = SoftwareMetadata.load_from_cache(ctx, harvester) + load_end = datetime.datetime.now() except Exception: # skip this harvester when the data is invalid + if prov_doc is not None: + remove_harvest_plugin_from_prov_doc(prov_doc, harvester) self.log.exception( f"### The data from the harvester {harvester} could not be loaded or is invalid, skipping it now." ) continue + if prov_doc is not None: + harvest_plugin = prov_doc.get_hermes_plugin("harvest", harvester) + harvest_command = prov_doc.get_hermes_command("harvest") + store_action = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [harvest_plugin.ref, hermes_cache.ref, harvest_command.ref] + ))[0] + stored_results = [ + result.ref for result in prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action.ref] + )) + ] + new_action = prov_doc.add_activity(data={ # load of new data + "schema:description": f"loads the data from {harvester} plugin", + "prov:wasAssociatedWith": [process_command.ref, hermes_cache.ref], + "prov:used": stored_results, + "prov:startedAtTime": load_start, + "prov:endedAtTime": load_end + }) + new_data = prov_doc.add_entity(data={ # new data to be merged + "@type": "schema:CreativeWork", + "schema:description": f"data loaded from {harvester} plugin", + "schema:text": str(metadata.compact()), + "prov:wasAttributedTo": [process_command.ref, hermes_cache.ref], + "prov:wasGeneratedBy": new_action.ref, + "prov:wasDerivedFrom": stored_results, + "prov:generatedAtTime": load_end + }) + if merged_any: + # One pass must have been completed successfully already. + new_action = prov_doc.add_activity(data={ + "schema:description": "merges the old data object with the new data", + "prov:used": [last_data.ref, new_data.ref, merged_strategies.ref], + "prov:wasInformedBy": [last_action.ref, new_action.ref, strategy_action.ref], + "prov:wasAssociatedWith": process_command.ref + }) # initial merge action of the merge + # set the starting objects of the merge + merge_doc.prov_objects = [new_action, new_data, last_data, merged_strategies, strategy_action] + self.log.info(f"### Merge data from {harvester} plugin") # merge data into the merge dict try: - merged_doc.update(metadata) + merge_start = datetime.datetime.now() + merge_doc.update(metadata) + merge_end = datetime.datetime.now() except Exception as e: + # TODO: Maybe this state is recoverable by starting over again and skipping this plugin. self.log.critical(f"### Merging the data from {harvester} plugin resulted in an error.", exc_info=True) raise RuntimeError(f"Merging the data from {harvester} plugin failed.") from e + + if prov_doc is not None: + if merged_any: + new_action["prov:startedAtTime"] = merge_start + new_action["prov:endedAtTime"] = merge_end + # set the last action and last data objects for next iteration + last_action = merge_doc.prov_objects[0] if merged_any else new_action + last_data = merge_doc.prov_objects[2] if merged_any else new_data merged_any = True # error if nothing was merged @@ -122,13 +399,28 @@ def __call__(self, args: argparse.Namespace) -> None: self.log.critical("No metadata has been merged, the loading of the data failed for all harvesters.") raise RuntimeError("No metadata has been merged.") - self.log.info("## Store processed metadata") - # store processed data - ctx.prepare_step("process") - with ctx["result"] as result_ctx: - result_ctx["codemeta"] = merged_doc.compact() - result_ctx["context"] = {"@context": merged_doc.full_context} - result_ctx["expanded"] = merged_doc.ld_value - ctx.finalize_step("process") + return last_action, last_data - ctx.finalize_step("harvest") + def load_prov_doc(self: Self) -> Optional[ld_prov_list]: + """ + Loads the provenance document of the harvest step. + + Returns: + ld_prov_list | None: The loaded provenance document or None if the load failed. + """ + # set up HermesContext + ctx = HermesCacheManager() + ctx.prepare_step("harvest") + with ctx["provenance"] as cache: + # try load + try: + return ld_prov_list.load_ld_prov_list(cache["result"]) + except Exception: + # log the warning and return None + self.log.warning( + "The provenance data from the harvest step could not be loaded. " + "Processing will proceed without collecting provenance data.", + exc_info=1 + ) + finally: + ctx.finalize_step("harvest") diff --git a/src/hermes/commands/process/invenio_merge.py b/src/hermes/commands/process/invenio_merge.py new file mode 100644 index 00000000..518364c5 --- /dev/null +++ b/src/hermes/commands/process/invenio_merge.py @@ -0,0 +1,97 @@ +# SPDX-FileCopyrightText: 2026 German Aerospace Center (DLR) +# +# SPDX-License-Identifier: Apache-2.0 + +# SPDX-FileContributor: Michael Fritzsche + +# flake8: noqa: C901 + +from typing import Union +from typing_extensions import Self + +from hermes.commands.base import HermesCommand +from hermes.model.merge.action import MergeAction +from hermes.model.merge.container import ld_merge_dict, ld_merge_list +from hermes.model.types import ld_dict, ld_list +from hermes.model.types.ld_container import BASIC_TYPE, TIME_TYPE +from hermes.model.types.ld_context import iri_map as iri +from .base import HermesProcessPlugin + + +class InvenioMerge(MergeAction): + """ :class:`MergeAction` providing a merge function that tries to conform with Invenios metadata restrictions. """ + def merge( + self: Self, + target: ld_merge_dict, + key: list[Union[str, int]], + value: Union[ld_merge_list, str], + update: Union[BASIC_TYPE, TIME_TYPE, ld_dict, ld_list] + ) -> ld_merge_list: + types = target.get("@type", []) + if ( + key[-1] == iri["schema:license"] and + (iri["schema:SoftwareSourceCode"] in types or iri["schema:SoftwareApplication"] in types) + ): + if len(value) == 1: + if isinstance(value[0], str) or ( + isinstance(value[0], (dict, ld_merge_dict)) and [*value[0].keys()] == ["@id"] + ): + if value != update: + target.reject(key[-1], update) + return value + if isinstance(update, ld_list) and len(update) == 1: + if isinstance(update[0], str) or ( + isinstance(update[0], (dict, ld_merge_dict)) and [*update[0].keys()] == ["@id"] + ): + target.replace(key[-1], value) + return update + target.reject(key[-1], update) + return value + if ( + (key[-1] == iri["schema:familyName"] and iri["schema:Person"] in types) or + (key[-1] == iri["schema:name"] and iri["schema:Person"] in types) or + (key[-1] == iri["schema:name"] and (iri["schema:SoftwareSourceCode"] in types or + iri["schema:SoftwareApplication"] in types)) + ): + if len(value) == 1: + if value != update: + target.reject(key[-1], update) + return value + if len(update) == 1: + target.replace(key[-1], value) + return update + if len(value) == len(update) == 0: + return value + target.reject(key[-1], update) + return value + if ( + (key[-1] == iri["schema:version"] or key[-1] == iri["schema:description"]) and + (iri["schema:SoftwareSourceCode"] in types or iri["schema:SoftwareApplication"] in types) + ): + if len(value) == 1: + if value != update: + target.reject(key[-1], update) + return value + if len(update) == 1: + target.replace(key[-1], value) + return update + if len(value) == 0 or len(update) == 0: + return [] + target.reject(key[-1], update) + return value + + +class InvenioProcessPlugin(HermesProcessPlugin): + def __call__(self, command: HermesCommand) -> dict[Union[str, None], dict[Union[str, None], MergeAction]]: + merger = InvenioMerge() + return { + iri["schema:SoftwareSourceCode"]: { + iri["schema:"+term]: merger for term in ["version", "name", "description", "license"] + }, + iri["schema:SoftwareApplication"]: { + iri["schema:"+term]: merger for term in ["version", "name", "description", "license"] + }, + iri["schema:Person"]: { + iri["schema:"+term]: merger for term in ["familyName", "name"] + } + } diff --git a/src/hermes/commands/report/__init__.py b/src/hermes/commands/report/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/src/hermes/commands/report/base.py b/src/hermes/commands/report/base.py new file mode 100644 index 00000000..7851c51e --- /dev/null +++ b/src/hermes/commands/report/base.py @@ -0,0 +1,471 @@ +# SPDX-FileCopyrightText: 2026 German Aerospace Center (DLR) +# +# SPDX-License-Identifier: Apache-2.0 + +# SPDX-FileContributor: Michael Fritzsche + +import argparse +from typing_extensions import Self + +from pydantic import BaseModel + +from hermes.commands.base import HermesCommand +from hermes.model.hermes_cache import HermesCacheManager +from hermes.model.provenance.ld_prov import ld_prov_list + + +class HermesReportSettings(BaseModel): + """ Configuration of the ``report`` command. """ + pass + + +class HermesReportCommand(HermesCommand): + """ + Generate a summarized provenance report for the steps chosen by the user. + + Attributes: + command_name (str): (class attribute) The name of the command. + settings_class (type): (class attribute) The settings class for general report settings. + """ + + command_name: str = "report" + settings_class: type = HermesReportSettings + + def init_command_parser(self: Self, command_parser: argparse.ArgumentParser) -> None: + """ + Add arguments for report command. + + Args: + command_parser (ArgumentParser): The used argument parser. + + Returns: + None: + """ + command_parser.add_argument( + "--steps", + nargs="*", + default=["harvest", "process", "curate", "deposit", "postprocess"], + choices=["harvest", "process", "curate", "deposit", "postprocess"], + help="Steps for which the report should be generated. Default is every step." + ) + + def __call__(self: Self, args: argparse.Namespace) -> None: + """ + Execute the hermes command `self`. + + Args: + args (Namespace): The namespace that was returned by the command line parser when reading the arguments. + + Returns: + None: + """ + print("\nProvenance report for HERMES:") + # print the report for every step + for step in args.steps: + # reset ld_prov_list because it is usually a singelton + ld_prov_list.INDICES = {} + # print the report + match step: + case "harvest": + self.report_harvest() + case "process": + self.report_process() + case "curate": + self.report_curate() + case "deposit": + self.report_deposit() + case "postprocess": + self.report_postprocess() + print("") + + def report_harvest(self: Self) -> None: + """ + Print the report for the harvest step. + + Returns: + None: + """ + print("- Harvest:") + # load provenance data or error out + ctx = HermesCacheManager() + ctx.prepare_step("harvest") + with ctx["provenance"] as cache: + try: + prov_doc = ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + print("No provenance data has been recorded so far.") + return + finally: + ctx.finalize_step("harvest") + # get basic hermes objects + harvest_base_plugin = prov_doc.get_hermes_base_plugin("harvest") + harvest_command = prov_doc.get_hermes_command("harvest") + hermes_cache = prov_doc.get_hermes_cache() + plugins = prov_doc.shallow_search(lambda node: ( + "prov:actedOnBehalfOf" in node and node["prov:actedOnBehalfOf"] == [harvest_base_plugin.ref] + )) + # for every plugin print the info on this plugins execution + for plugin in plugins: + # print basic info on the plugin + print( + f" - Plugin {plugin['@id'][24:]} ({plugin['schema:name'][0]}, version " + f"{vers if (vers := plugin.get('schema:softwareVersion', False)) else 'N/A'})" + ) + # print every source loaded by the plugin + print(" - Loaded data from:") + for load_action in prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [harvest_base_plugin.ref, plugin.ref] + )): + id_of_source = load_action["prov:used"][0]["@id"] + source = prov_doc.shallow_search(lambda node: ("@id" in node and node["@id"] == id_of_source))[0] + print( + f" - {source['schema:url'][0]} (at {load_action['prov:startedAtTime'][0]}, took " + + f"{load_action['prov:endedAtTime'][0]-load_action['prov:startedAtTime'][0]})" + ) + # print infos on the result + store_action = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [plugin.ref, hermes_cache.ref, harvest_command.ref] + ))[0] + print( + f" - Results stored (at {store_action['prov:startedAtTime'][0]}, took " + f"{store_action['prov:endedAtTime'][0]-store_action['prov:startedAtTime'][0]}) in:" + ) + for result in prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action.ref] + )): + print(f" - {result['schema:url'][0]} ({result['schema:description'][0].split(' ')[1]})") + + def report_process(self: Self) -> None: + """ + Print the report for the process step. + + Returns: + None: + """ + print("- Process:") + # load provenance data or error out + ctx = HermesCacheManager() + ctx.prepare_step("process") + with ctx["provenance"] as cache: + try: + prov_doc = ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + print("No provenance data has been recorded so far.") + return + finally: + ctx.finalize_step("process") + # get basic hermes objects + process_base_plugin = prov_doc.get_hermes_base_plugin("process") + process_command = prov_doc.get_hermes_command("process") + hermes_cache = prov_doc.get_hermes_cache() + plugins = prov_doc.shallow_search(lambda node: ( + "prov:actedOnBehalfOf" in node and node["prov:actedOnBehalfOf"] == [process_base_plugin.ref] + )) + # print info on all plugins and their strategy generation + for plugin in plugins: + print( + f" - Plugin {plugin['@id'][24:]} ({plugin['schema:name'][0]}, version " + f"{vers if (vers := plugin.get('schema:softwareVersion', False)) else 'N/A'}):" + ) + strategy_generation = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and node["prov:wasAssociatedWith"] == [plugin.ref] + ))[0] + print( + f" - Generated strategies at {strategy_generation['prov:startedAtTime'][0]} took " + f"{strategy_generation['prov:endedAtTime'][0]-strategy_generation['prov:startedAtTime'][0]}" + ) + load_actions = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [hermes_cache.ref, process_command.ref] and + "prov:used" in node and + len(node["prov:used"]) == 3 + )) + # print info on the data loaded that was merged + for index, load_action in enumerate(sorted(load_actions, key=lambda it: it["prov:startedAtTime"][0]), start=1): + print( + f" - In load {index} loaded (at {load_action['prov:startedAtTime'][0]}, took" + f" {load_action['prov:endedAtTime'][0]-load_action['prov:startedAtTime'][0]}" + ", may have been overwritten) from:" + ) + loaded = [item["@id"] for item in load_action["prov:used"]] + sources = prov_doc.shallow_search(lambda node: ("@id" in node and node["@id"] in loaded)) + for source in sources: + print(f" - {source['schema:url'][0]}") + bigest_mergers = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [process_command.ref] and + "prov:wasInformedBy" in node and + len(node["prov:wasInformedBy"]) == 3 + )) + # print info on the merges + for index, merger in enumerate(sorted(bigest_mergers, key=lambda it: it["prov:startedAtTime"][0]), start=1): + if index == 1: + merged = "merged data from load 1 with data of load 2" + else: + merged = f"merged data from load {index + 1} with old results" + print( + f" - Merge {index} {merged} at {merger['prov:startedAtTime'][0]} took " + f"{merger['prov:endedAtTime'][0]-merger['prov:startedAtTime'][0]}" + ) + write_action = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [hermes_cache.ref, process_command.ref] and + "prov:used" in node and + len(node["prov:used"]) == 1 + ))[0] + stored_objects = prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [write_action.ref] + )) + # print info on the stored data + print( + f" - Results stored (at {write_action['prov:startedAtTime'][0]} took " + f"{write_action['prov:endedAtTime'][0]-write_action['prov:startedAtTime'][0]}) in:" + ) + for res in stored_objects: + print(f" - {res['schema:url'][0]} ({res['schema:description'][0].split(' ')[1]})") + + def report_curate(self: Self) -> None: + """ + Print the report for the curate step. + + Returns: + None: + """ + print("- Curate:") + # load provenance data or error out + ctx = HermesCacheManager() + ctx.prepare_step("curate") + with ctx["provenance"] as cache: + try: + prov_doc = ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + print("No provenance data has been recorded so far.") + return + finally: + ctx.finalize_step("curate") + # get basic hermes objects + curate_base_plugin = prov_doc.get_hermes_base_plugin("curate") + process_command = prov_doc.get_hermes_command("process") + hermes_cache = prov_doc.get_hermes_cache() + # print info on the used plugin + curate_plugin = prov_doc.shallow_search(lambda node: ( + "prov:actedOnBehalfOf" in node and node["prov:actedOnBehalfOf"] == [curate_base_plugin.ref] + ))[0] + print( + f" - Plugin used:\n - {curate_plugin['@id'][23:]} ({curate_plugin['schema:name'][0]}, version " + f"{vers if (vers := curate_plugin.get('schema:softwareVersion', False)) else 'N/A'})" + ) + # get objects that contain info on the curation + store_action_of_process = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [process_command.ref, hermes_cache.ref] and + "prov:wasInformedBy" in node + ))[0] + stored_results_of_process = prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action_of_process.ref] + )) + load_action = prov_doc.shallow_search(lambda node: ( + "prov:used" in node and node["prov:used"] == [res.ref for res in stored_results_of_process] + ))[0] + curate_activity = prov_doc.shallow_search(lambda node: ( + "prov:wasInfluencedBy" in node and node["prov:wasInfluencedBy"] == [curate_plugin.ref] + ))[0] + results = prov_doc.shallow_search(lambda node: ( + "prov:wasDerivedFrom" in node and node["prov:wasDerivedFrom"] == [curate_activity.ref] + )) + write = prov_doc.shallow_search(lambda node: ( + "prov:used" in node and node["prov:used"] == [curate_activity.ref] + ))[0] + # print curation info + print( + f" - Time consumed:\n - Curation at ~{load_action['prov:endedAtTime'][0]} took" + f" ~{curate_activity['prov:generatedAtTime'][0]-load_action['prov:endedAtTime'][0]}" + ) + print( + f" - Uncurated metadata loaded (at {load_action['prov:startedAtTime'][0]}" + f", took {load_action['prov:endedAtTime'][0]-load_action['prov:startedAtTime'][0]}" + ", may have been overwritten) from:" + ) + for source in stored_results_of_process: + print(4*" " + f"- {source['schema:url'][0]} ({source['schema:description'][0].split(' ')[1]})") + print( + f" - Curated metadata stored (at {write['prov:startedAtTime'][0]}, took " + f"{write['prov:endedAtTime'][0]-write['prov:startedAtTime'][0]}) in:" + ) + for result in results: + print(f" - {result['schema:url'][0]} ({result['schema:description'][0].split(' ')[1]})") + + def report_deposit(self: Self) -> None: + """ + Print the report for the deposit step. + + Returns: + None: + """ + print("- Deposit:") + # load provenance data or error out + ctx = HermesCacheManager() + ctx.prepare_step("deposit") + with ctx["provenance"] as cache: + try: + prov_doc = ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + print("No provenance data has been recorded so far.") + return + finally: + ctx.finalize_step("deposit") + # get basic hermes objects + deposit_base_plugin = prov_doc.get_hermes_base_plugin("deposit") + curate_command = prov_doc.get_hermes_command("curate") + hermes_cache = prov_doc.get_hermes_cache() + # print info on the plugin + deposit_plugin = prov_doc.shallow_search(lambda node: ( + "prov:actedOnBehalfOf" in node and node["prov:actedOnBehalfOf"] == [deposit_base_plugin.ref] + ))[0] + print( + f" - Plugin used:\n - {deposit_plugin['@id'][24:]} ({deposit_plugin['schema:name'][0]}, version " + f"{vers if (vers := deposit_plugin.get('schema:softwareVersion', False)) else 'N/A'})" + ) + # get objects containing info on map and update of the metadata + store_action_of_curate = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [curate_command.ref, hermes_cache.ref] and + "prov:used" in node and + len(node["prov:used"]) == 1 + ))[0] + stored_results_of_curate = prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_action_of_curate.ref] + )) + load_action = prov_doc.shallow_search(lambda node: ( + "prov:used" in node and node["prov:used"] == [res.ref for res in stored_results_of_curate] + ))[0] + mapped_metadata = prov_doc.shallow_search(lambda node: ( + "prov:wasAttributedTo" in node and node["prov:wasAttributedTo"] == [deposit_plugin.ref] + ))[0] + store_mapped = prov_doc.shallow_search(lambda node: ( + "prov:used" in node and node["prov:used"] == [mapped_metadata.ref] + ))[0] + result_mapped = prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [store_mapped.ref] + ))[0] + updated_metadata = prov_doc.shallow_search(lambda node: ( + "prov:wasInfluencedBy" in node and node["prov:wasInfluencedBy"] == [deposit_plugin.ref] + ))[0] + result_updated = prov_doc.shallow_search(lambda node: ( + "prov:wasDerivedFrom" in node and node["prov:wasDerivedFrom"] == [updated_metadata.ref] + ))[0] + store_updated = prov_doc.shallow_search(lambda node: ( + "prov:used" in node and node["prov:used"] == [updated_metadata.ref] + ))[0] + map_action = prov_doc.shallow_search(lambda node: ( + "@id" in node and node["@id"] == mapped_metadata["prov:wasGeneratedBy"][0]["@id"] + ))[0] + # print general info and info on map as well as update of the metadata + print( + " - Time consumed:\n" + f" - Preparation at ~{load_action['prov:endedAtTime'][0]} took ~" + f"{map_action['prov:startedAtTime'][0]-load_action['prov:endedAtTime'][0]}\n" + f" - Mapping at ~{map_action['prov:startedAtTime'][0]} took ~" + f"{map_action['prov:endedAtTime'][0]-map_action['prov:startedAtTime'][0]}\n" + f" - Creating new or initial version and updating metadata at ~{store_mapped['prov:endedAtTime'][0]}" + f" took ~{updated_metadata['prov:generatedAtTime'][0]-store_mapped['prov:endedAtTime'][0]}\n" + f" - Deletion of artifacts, upload of artifacts and publication at" + f" ~{store_updated['prov:endedAtTime'][0]} took N/A\n" + f" - Curated metadata loaded (at {load_action['prov:startedAtTime'][0]}" + f", took {load_action['prov:endedAtTime'][0]-load_action['prov:startedAtTime'][0]}" + ", may have been overwritten) from:" + ) + for source in stored_results_of_curate: + print(4*" " + f"- {source['schema:url'][0]} ({source['schema:description'][0].split(' ')[1]})") + # print info on the store of the mapped as well as updated metadata + print( + f" - Metadata mapped for deposit stored (at {store_mapped['prov:startedAtTime'][0]}, took " + f"{store_mapped['prov:endedAtTime'][0]-store_mapped['prov:startedAtTime'][0]}) in:\n" + f" - {result_mapped['schema:url'][0]}\n" + f" - Metadata updated after deposit stored (at {store_updated['prov:startedAtTime'][0]}, took " + f"{store_updated['prov:endedAtTime'][0]-store_updated['prov:startedAtTime'][0]}) in:\n" + f" - {result_updated['schema:url'][0]}" + ) + + def report_postprocess(self: Self) -> None: + """ + Print the report for the harvest step. + + Returns: + None: + """ + print("- Postprocess:") + # load provenance data or error out + ctx = HermesCacheManager() + ctx.prepare_step("postprocess") + with ctx["provenance"] as cache: + try: + prov_doc = ld_prov_list.load_ld_prov_list(cache["result"]) + except KeyError: + print("No provenance data has been recorded so far.") + return + finally: + ctx.finalize_step("postprocess") + # get basic hermes objects + cache = prov_doc.get_hermes_cache() + command = prov_doc.get_hermes_command("postprocess") + base_plugin = prov_doc.get_hermes_base_plugin("postprocess") + # print info on the plugin + plugin = prov_doc.shallow_search(lambda node: ( + "prov:actedOnBehalfOf" in node and node["prov:actedOnBehalfOf"] == [base_plugin.ref] + ))[0] + print( + f" - Plugin used:\n - {plugin['@id'][28:]} ({plugin['schema:name'][0]}, version " + f"{vers if (vers := plugin.get('schema:softwareVersion', False)) else 'N/A'})" + ) + cache_loads = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [plugin.ref, base_plugin.ref, command.ref, cache.ref] + )) + # print info on the loads from cache + print(" - Used deposit results:") + for index, cache_load in enumerate(cache_loads, start=1): + source_id = cache_load["prov:used"][0]["@id"] + source = prov_doc.shallow_search(lambda node: ("@id" in node and node["@id"] == source_id))[0] + print( + f" - Load {index} at {cache_load['prov:startedAtTime'][0]} took " + f"{cache_load['prov:endedAtTime'][0]-cache_load['prov:startedAtTime'][0]} from:\n" + f" - {source['schema:url'][0]}" + ) + io_ops = prov_doc.shallow_search(lambda node: ( + "prov:wasAssociatedWith" in node and + node["prov:wasAssociatedWith"] == [plugin.ref, base_plugin.ref, command.ref] + )) + # sort io operations into load and write operations + loads, writes = [], [] + for io_op in io_ops: + used = io_op["prov:used"][0]["@id"] + if "prov:wasDerivedFrom" in prov_doc.shallow_search( + lambda node: ("@id" in node and node["@id"] == used) + )[0]: + writes.append(io_op) + else: + loads.append(io_op) + # print info on general loads of the plugin + print(" - Loaded data from:") + for index, load in enumerate(loads): + source_id = load["prov:used"][0]["@id"] + source = prov_doc.shallow_search(lambda node: ("@id" in node and node["@id"] == source_id))[0] + print( + f" - Load {index} at {load['prov:startedAtTime'][0]} took " + f"{load['prov:endedAtTime'][0]-load['prov:startedAtTime'][0]} from:\n" + f" - {source['schema:url'][0]}" + ) + # print info on general writes of the plugin + print(" - Written data to:") + for index, write in enumerate(writes): + target = prov_doc.shallow_search(lambda node: ( + "prov:wasGeneratedBy" in node and node["prov:wasGeneratedBy"] == [write.ref] + ))[0] + print( + f" - Write {index} at {write['prov:startedAtTime'][0]} took " + f"{write['prov:endedAtTime'][0]-write['prov:startedAtTime'][0]} from:\n" + f" - {target['schema:url'][0]}" + ) diff --git a/src/hermes/model/error.py b/src/hermes/model/error.py index f0d8a009..2a86dc5d 100644 --- a/src/hermes/model/error.py +++ b/src/hermes/model/error.py @@ -47,9 +47,9 @@ class HermesMergeError(Exception): This exception should be raised when there is an error during a merge / set operation. Attributes: - path (list[str | int]): The path where the merge error occured. + path (list[str | int]): The path where the merge error occurred. old_value (Any): Old value that was stored at `path`. - new_value (Any): New value that was to be assinged. + new_value (Any): New value that was to be assigned . tag: Tag data for the new value. """ def __init__(self, path: list[Union[str, int]], old_value: Any, new_value: Any, **kwargs) -> None: @@ -57,9 +57,9 @@ def __init__(self, path: list[Union[str, int]], old_value: Any, new_value: Any, Create a new merge incident. Args: - path (list[str | int]): The path where the merge error occured. + path (list[str | int]): The path where the merge error occurred. old_value (Any): Old value that was stored at `path`. - new_value (Any): New value that was to be assinged. + new_value (Any): New value that was to be assigned . kwargs: Tag data for the new value. Returns: diff --git a/src/hermes/model/merge/action.py b/src/hermes/model/merge/action.py index f2cfc7b3..c95b7ef3 100644 --- a/src/hermes/model/merge/action.py +++ b/src/hermes/model/merge/action.py @@ -49,6 +49,16 @@ def merge( """ raise NotImplementedError() + def __repr__(self) -> str: + """ + A generic stringify method for MergeActions. + Please overwrite this method if your MergeAction should be represented differently in the provenance data. + (I.e. if not all important attributes are recorded or some attributes string representation is not adequat.) + """ + if self.__dict__: + return f"{self.__module__}.{self.__class__.__qualname__} with attributes {str(self.__dict__)}" + return f"{self.__module__}.{self.__class__.__qualname__}" + class Reject(MergeAction): """ :class:`MergeAction` providing a merge function for rejecting the incoming item. """ @@ -209,6 +219,10 @@ def merge( return value + def __repr__(self): + return f"{self.__module__}.{self.__class__.__qualname__} with attributes " \ + f"{{'match': {self.match.__module__}.{self.match.__qualname__}, 'reject_incoming': {self.reject_incoming}}}" + class MergeSet(MergeAction): """ @@ -268,13 +282,19 @@ def merge( elif isinstance(item, ld_list) and isinstance(update_item, ld_list): self.merge(target, [*key, index], item, update_item) elif isinstance(item, (ld_dict, ld_list)) or isinstance(update_item, (ld_dict, ld_list)): - """ FIXME: log error """ + """ + FIXME: log error/ warning that merge of items at... could not be merged and will be skipped + """ break else: value.append(update_item) # Return the merged values. return value + def __repr__(self): + return f"{self.__module__}.{self.__class__.__qualname__} with attributes " \ + f"{{'match': {self.match.__module__}.{self.match.__qualname__}}}" + class IdMerge(MergeAction): """ :class:`MergeAction` providing a merge function for merging ids, i.e. error if not equals else do nothing. """ diff --git a/src/hermes/model/merge/container.py b/src/hermes/model/merge/container.py index c837295f..c68e96cb 100644 --- a/src/hermes/model/merge/container.py +++ b/src/hermes/model/merge/container.py @@ -7,9 +7,11 @@ from __future__ import annotations +import datetime from typing import TYPE_CHECKING, Any, Callable, Optional, Union from typing_extensions import Self +from hermes.model.provenance.ld_prov import ld_prov_list from hermes.model.types import ld_container, ld_context, ld_dict, ld_list from hermes.model.types.ld_container import ( BASIC_TYPE, EXPANDED_JSON_LD_VALUE, JSON_LD_CONTEXT_DICT, JSON_LD_VALUE, TIME_TYPE @@ -50,6 +52,8 @@ def _to_native_python( if isinstance(value, ld_dict) and not isinstance(value, ld_merge_dict): value = ld_merge_dict( value.ld_value, + self.prov_doc, + self.prov_objects, parent=value.parent, key=value.key, index=value.index, @@ -60,6 +64,8 @@ def _to_native_python( if isinstance(value, ld_list) and not isinstance(value, ld_merge_list): value = ld_merge_list( value.ld_value, + self.prov_doc, + self.prov_objects, parent=value.parent, key=value.key, index=value.index, @@ -82,6 +88,8 @@ class ld_merge_list(_ld_merge_container, ld_list): def __init__( self: "ld_merge_list", data: Union[list[str], list[dict[str, EXPANDED_JSON_LD_VALUE]]], + prov_doc: ld_prov_list = None, + prov_objects: list[ld_dict] = 3*[None], *, parent: Optional[ld_container] = None, key: Optional[str] = None, @@ -108,6 +116,8 @@ def __init__( super().__init__(data, parent=parent, key=key, index=index, context=context) self.strategies = strategies + self.prov_doc = prov_doc + self.prov_objects = prov_objects class ld_merge_dict(_ld_merge_container, ld_dict): @@ -123,6 +133,8 @@ class ld_merge_dict(_ld_merge_container, ld_dict): def __init__( self: Self, data: list[dict[str, EXPANDED_JSON_LD_VALUE]], + prov_doc: ld_prov_list = None, + prov_objects: list[ld_dict] = 3*[None], *, parent: Optional[Union[ld_dict, ld_list]] = None, key: Optional[str] = None, @@ -154,6 +166,8 @@ def __init__( # add strategies self.strategies = strategies + self.prov_doc = prov_doc + self.prov_objects = prov_objects def update_context( self: Self, other_context: Union[list[Union[str, JSON_LD_CONTEXT_DICT]], None] @@ -224,10 +238,50 @@ def __setitem__(self: Self, key: str, value: Union[JSON_LD_VALUE, BASIC_TYPE, TI ``self[key]``. """ # create the new item if self[key] and value have to be merged. + merge_start = datetime.datetime.now() if key in self: - value = self._merge_item(key, value) + if self.prov_objects[0] is not None: + last_merged_data = self.prov_objects[2] + merge_activity, value = self._merge_item(key, value) + if self.prov_objects[0] is not None: + create_new_merged_data = last_merged_data is self.prov_objects[2] + elif self.prov_objects[0] is not None: + merge_activity = self.prov_doc.add_activity(data={ + "schema:name": f"merge values at {str(self.path+[key])}", + "schema:description": f"Inserting value in the second 'used' value at {str(self.path+[key])} into the " + "first 'used' value at the same point, no merger needed.", + "prov:used": {"@list": [ + self.prov_objects[2].ref, + self.prov_objects[1].ref, + self.prov_objects[3].ref + ]}, + "prov:wasInformedBy": {"@list": [self.prov_objects[0].ref, self.prov_objects[4].ref]} + }) + create_new_merged_data = True # update the entry of self[key] super().__setitem__(key, value) + merge_end = datetime.datetime.now() + if self.prov_objects[0] is None: + return + if merge_activity is not None: + merge_activity["prov:startedAtTime"] = merge_start + merge_activity["prov:endedAtTime"] = merge_end + self.prov_objects[0] = merge_activity + if create_new_merged_data: + outer_most_parent = self + while outer_most_parent.parent is not None: + outer_most_parent = outer_most_parent.parent + self.prov_objects[2] = self.prov_doc.add_entity(data={ + "@type": "schema:CreativeWork", + "schema:description": f"software metadata after merge of values at {str(self.path+[key])}", + "schema:text": str(outer_most_parent.compact()), + "prov:wasAttributedTo": self.prov_doc.get_hermes_command("process").ref, + "prov:wasGeneratedBy": merge_activity.ref, + "prov:wasDerivedFrom": {"@list": [self.prov_objects[1].ref, self.prov_objects[2].ref]}, + "prov:generatedAtTime": merge_end + }) + else: + self.prov_objects[2]["prov:wasGeneratedBy"].append(merge_activity.ref) def match( self: Self, @@ -274,14 +328,36 @@ def _merge_item( # search for all applicable strategies strategy = {**self.strategies.get(None, {})} ld_types = self.data_dict.get('@type', []) + type_of_used_strategy = None + key_of_used_strategy = None for ld_type in ld_types: strategy.update(self.strategies.get(ld_type, {})) + if key in self.strategies.get(ld_type, {}): + type_of_used_strategy = ld_type + key_of_used_strategy = key # choose one merge strategy and return the item returned by following the merge startegy merger = strategy.get(key, strategy.get(None, None)) if merger is None: raise MergeError(f"Can't merge, no strategy found for key '{key}'.") - return merger.merge(self, [*self.path, key], self[key], value) + if self.prov_objects[0] is not None: + merge_activity = self.prov_doc.add_activity(data={ + "schema:name": f"merge values at {str(self.path+[key])}", + "schema:description": f"Merge value in the second 'used' value at {str(self.path+[key])} into the " + f"first 'used' value at the same point using the merger {merger} for type " + f"{type_of_used_strategy} and key {key_of_used_strategy}", + "prov:wasAssociatedWith": self.prov_doc.get_hermes_command("process").ref, + "prov:used": {"@list": [ + self.prov_objects[2].ref, + self.prov_objects[1].ref, + self.prov_objects[3].ref + ]}, + "prov:wasInformedBy": {"@list": [self.prov_objects[0].ref, self.prov_objects[4].ref]} + }) + self.prov_objects[0] = merge_activity + else: + merge_activity = None + return merge_activity, merger.merge(self, [*self.path, key], self[key], value) def _add_related( self: Self, rel: str, key: str, value: Union[BASIC_TYPE, TIME_TYPE, ld_dict, ld_list] @@ -297,7 +373,6 @@ def _add_related( Returns: None: """ - # FIXME: key not only string # make sure appending is possible self.emplace(rel) # append the new entry @@ -316,7 +391,6 @@ def reject(self: Self, key: str, value: Union[BASIC_TYPE, TIME_TYPE, ld_dict, ld Returns: None: """ - # FIXME: key not only string self._add_related("hermes-rt:reject", key, value) def replace(self: Self, key: str, value: Union[BASIC_TYPE, TIME_TYPE, ld_dict, ld_list]) -> None: @@ -332,5 +406,4 @@ def replace(self: Self, key: str, value: Union[BASIC_TYPE, TIME_TYPE, ld_dict, l Returns: None: """ - # FIXME: key not only string self._add_related("hermes-rt:replace", key, value) diff --git a/src/hermes/model/provenance/ld_prov.py b/src/hermes/model/provenance/ld_prov.py new file mode 100644 index 00000000..eda1bbc1 --- /dev/null +++ b/src/hermes/model/provenance/ld_prov.py @@ -0,0 +1,451 @@ +# SPDX-FileCopyrightText: 2026 German Aerospace Center (DLR) +# +# SPDX-License-Identifier: Apache-2.0 + +# SPDX-FileContributor: Michael Fritzsche + +from importlib.metadata import metadata +from typing import Any, Callable, Optional, Union +from typing_extensions import Self + +from hermes import utils +from hermes.commands.base import HermesCommand, HermesPlugin +from hermes.model.types import ld_dict, ld_list +from hermes.model.types.ld_container import EXPANDED_JSON_LD_VALUE, JSON_LD_CONTEXT_DICT, JSON_LD_VALUE +from hermes.model.types.ld_context import ALL_CONTEXTS, iri_map + + +class ld_prov_list(ld_list): + """ + ld_list with special features for internal provenance collection. + + Attributes: + NODE_IRI_FORMAT (str): (class attribute) The id format of normal nodes. + HERMES_ID (str): (class attribute) The id of the hermes agent. + HERMES_CACHE_ID (str): (class attribute) The id of the hermes cache. + HERMES_COMMAND_ID_FORMAT (str): (class attribute) The id format of hermes commands. + HERMES_PLUGIN_ID_FORMAT (str): (class attribute) The id format of hermes plugins. + HERMES_BASE_PLUGIN_ID_FORMAT (str): (class attribute) The id format of hermes base plugins. + PROV_DOC_IRI (str): (class attribute) The JSON-LD type of the prov_doc itself. + INDICES (dict[str, int]): (class attribute) The counters of the different types of nodes. + """ + NODE_IRI_FORMAT: str = "_:{type}/{index}" + HERMES_ID: str = f"https://doi.org/{utils.hermes_doi}" + HERMES_CACHE_ID: str = "_:hermes/cache" + HERMES_COMMAND_ID_FORMAT: str = "_:hermes/command/{step}" + HERMES_PLUGIN_ID_FORMAT: str = "_:hermes/plugin/{step}/{name}" + HERMES_BASE_PLUGIN_ID_FORMAT: str = "_:hermes/base_plugin/{step}" + PROV_DOC_IRI: str = iri_map['hermes-rt', "graph"] + INDICES: dict[str, int] = {} + + def __init__( + self: Self, + data: EXPANDED_JSON_LD_VALUE = [{"@graph": []}], + *, + parent: Optional[Union[ld_dict, ld_list]] = None, + key: Optional[str] = PROV_DOC_IRI, + index: Optional[int] = None, + context: Optional[list[Union[str, JSON_LD_CONTEXT_DICT]]] = ALL_CONTEXTS + ) -> None: + """ + Create a new instance of an ld_prov_list, should not be used. + Use :meth:`ld_prov_list.load_ld_prov_list` instead. + See also :meth:`ld_list.__init__`. + + Args: + data (EXPANDED_JSON_LD_VALUE): The expanded json-ld data that represents the list, default is an empty graph + parent (ld_dict | ld_list | None): parent node of this container. + key (str | None): key into the parent container. + index (int | None): index into the parent container. + context (list[str | JSON_LD_CONTEXT_DICT] | None): local context for this container. + + Returns: + None: + """ + super().__init__(data, parent=parent, key=key, index=index, context=context) + + @classmethod + def load_ld_prov_list(cls: type[Self], data: EXPANDED_JSON_LD_VALUE) -> "ld_prov_list": + """ + Create a new instance of an ld_merge_dict. See also :meth:`ld_dict.__init__`. + + Args: + data (EXPANDED_JSON_LD_VALUE): The expanded json-ld data from which an ld_prov_list is restored. + + Returns: + ld_prov_list: The ld_prov_list loaded from the provided data. + + Raises: + RuntimeError: If an ld_prov_list has/ had been loaded before. + """ + # check if an ld_prov_list has/ had been loaded before + if cls.INDICES != {}: + raise RuntimeError("Only zero or one objects of class 'ld_prov_list' may exist at every point in time.") + # create ld_prov_list from the data + prov_list = cls.from_list( + data[0]["@graph"], key=cls.PROV_DOC_IRI, context=ALL_CONTEXTS, container_type="@graph" + ) + # initialize counters for different node types + for item in prov_list: + if not ("@id" in item and item["@id"].startswith("_:")): + continue + item_id = item["@id"][2:].split("/") + if not (len(item_id) == 2 and item_id[1].isnumeric()): + continue + if cls.INDICES.get(item_id[0], 0) < int(item_id[1]): + cls.INDICES[item_id[0]] = int(item_id[1]) + return prov_list + + def next_node_iri(self: Self, type: str) -> str: + """ + Create an iri for a new node of the given type. + + Args: + type (str): The type of the new node + + Returns: + str: The generated iri. + """ + # update counter for the given type + if type not in ld_prov_list.INDICES: + ld_prov_list.INDICES[type] = 0 + ld_prov_list.INDICES[type] += 1 + # generate and return the iri + return self.NODE_IRI_FORMAT.format(type=type, index=ld_prov_list.INDICES[type]) + + def add_activity(self: Self, *, data: JSON_LD_VALUE = {}) -> ld_dict: + """ + Add a new provenance activity to the ld_prov_list using the provided additional data. + + Hint: If no id was specified, one will be generated. Additionaly the types 'prov:Activity' and + 'schema:Action' will be added. + + Args: + data (JSON_LD_VALUE): The additional data for the activity. + + Returns: + ld_dict: The provenance activity as an ld_dict (can be used to update the data in the ld_prov_list). + """ + # add and get the object + self.append(data) + activity = self[-1] + # add the additional types + if "@type" not in data: + activity["@type"] = ["prov:Activity", "schema:Action"] + else: + activity["@type"].extend(["prov:Activity", "schema:Action"]) + # add an id if necessary + if "@id" not in data: + activity["@id"] = self.next_node_iri("Activity") + # return the object + return activity + + def add_agent(self: Self, *, data: JSON_LD_VALUE = {}) -> ld_dict: + """ + Add a new provenance agent to the ld_prov_list using the provided additional data. + + Hint: If no id was specified, one will be generated. Additionaly the types 'prov:Agent' and + 'schema:SoftwareApplication' will be added. + + Args: + data (JSON_LD_VALUE): The additional data for the agent. + + Returns: + ld_dict: The provenance agent as an ld_dict (can be used to update the data in the ld_prov_list). + """ + # add and get the object + self.append(data) + agent = self[-1] + # add the additional types + if "@type" not in data: + agent["@type"] = ["prov:Agent", "schema:SoftwareApplication"] + else: + agent["@type"].extend(["prov:Agent", "schema:SoftwareApplication"]) + # add an id if necessary + if "@id" not in data: + agent["@id"] = self.next_node_iri("Agent") + # return the object + return agent + + def add_entity(self: Self, *, data: JSON_LD_VALUE = {}) -> ld_dict: + """ + Add a new provenance entity to the ld_prov_list using the provided additional data. + + Hint: If no id was specified, one will be generated. Additionaly the types 'prov:Entity' and + 'schema:Thing' will be added. + + Args: + data (JSON_LD_VALUE): The additional data for the entity. + + Returns: + ld_dict: The provenance entity as an ld_dict (can be used to update the data in the ld_prov_list). + """ + # add and get the object + self.append(data) + entity = self[-1] + # add the additional types + if "@type" not in data: + entity["@type"] = ["prov:Entity", "schema:Thing"] + else: + entity["@type"].extend(["prov:Entity", "schema:Thing"]) + # add an id if necessary + if "@id" not in data: + entity["@id"] = self.next_node_iri("Entity") + # return the object + return entity + + def init_hermes_agents(self: Self) -> None: + """ + Initialize the hermes agents for provenance collection. + + Returns: + None: + """ + # add an agent for both hermes itself and the hermes cache + hermes = self.add_agent(data={ + "@id": ld_prov_list.HERMES_ID, + "@type": "schema:SoftwareApplication", + "schema:name": utils.hermes_name, + "schema:version": utils.hermes_version, + "schema:url": [*set(utils.hermes_urls.values())] + }) + self.add_agent(data={ + "@id": ld_prov_list.HERMES_CACHE_ID, + "@type": "schema:SoftwareApplication", + "schema:name": utils.hermes_name + " cache", + "schema:version": utils.hermes_version, + "prov:actedOnBehalfOf": hermes.ref + }) + # add the agents for each command and base plugin + for step in ["harvest", "process", "curate", "deposit", "postprocess"]: + command = self.add_agent(data={ + "@id": ld_prov_list.HERMES_COMMAND_ID_FORMAT.format(step=step), + "@type": "schema:SoftwareApplication", + "schema:name": f"{utils.hermes_name} {step} command", + "schema:version": utils.hermes_version, + "prov:actedOnBehalfOf": hermes.ref + }) + self.add_agent(data={ + "@id": ld_prov_list.HERMES_BASE_PLUGIN_ID_FORMAT.format(step=step), + "@type": "schema:SoftwareApplication", + "schema:name": f"{utils.hermes_name} {step} base plugin", + "schema:version": utils.hermes_version, + "prov:actedOnBehalfOf": command.ref + }) + + def add_hermes_settings(self: Self, command: HermesCommand) -> None: + """ + Add general settings of a hermes command run from the command object. + + Args: + command (HermesCommand): The command object containing information on the run. + + Returns: + None: + """ + # add basic settings + hermes = self.get_hermes() + hermes.emplace("schema:supportingData") + hermes["schema:supportingData"].append({ + "@type": "schema:DataFeed", + "schema:dataFeedElement": [ + { + "@type": "schema:DataFeedItem", + "schema:name": name, + "schema:item": [ + { + "@type": "schema:Item", + "schema:description": value + } + ], + "schema:description": "setting provided by command line (or its default value)" + } + for name, value in [ + ("path", command.args.path.absolute().as_uri()), + ("config", command.args.config.absolute().as_uri()), + ("options", str(command.args.options)) + ] + ], + "schema:description": f"options for run {len(hermes['schema:supportingData']) + 1} of some hermes step" + }) + # add command specific settings + for name, values in command.root_settings.model_dump(mode="json").items(): + if not isinstance(values, list): + values = [values] + hermes["schema:supportingData"][-1]["schema:dataFeedElement"].append({ + "@type": "schema:DataFeedItem", + "schema:name": name, + "schema:item": [ + { + "@type": "schema:Item", + "schema:description": value + } + for value in values + ], + "schema:description": "setting loaded from the config file" + }) + + def add_settings_to_command(self: Self, step: str, command: HermesCommand) -> None: + """ + Add settings specific to the ran command from the command object. + :meth:`ld_prov_list.add_hermes_settings` must be run before this function. + + Args: + step (str): The step of the settings should be recorded for. + command (HermesCommand): The command object containing information on the run. + + Returns: + None: + """ + # add basics + command_prov = self.get_hermes_command(step) + command_prov.emplace("schema:supportingData") + command_prov["schema:supportingData"].append({ + "@type": "schema:DataFeed", + "schema:dataFeedElement": [], + "schema:description": f"options for run {len(command_prov['schema:supportingData']) + 1} of step {step} out" + f" of {len(self.get_hermes()['schema:supportingData'])} runs of some hermes step" + }) # Needs add_hermes_settings to be called before add_settings_to_command is called! + # add specific settings to the command + for name, values in command.settings.model_dump(mode="json").items(): + if not isinstance(values, list): + values = [values] + command_prov["schema:supportingData"][-1]["schema:dataFeedElement"].append({ + "@type": "schema:DataFeedItem", + "schema:name": name, + "schema:item": [ + { + "@type": "schema:Item", + "schema:description": value + } + for value in values + ] + }) + + def add_hermes_plugin(self: Self, step: str, name: str, plugin: HermesPlugin, command: HermesCommand) -> ld_dict: + """ + Add a new hermes plugin to the ld_prov_list using the provided additional data. + + Args: + step (str): The step of the plugin. + name (str): The name of the plugin. + plugin (HermesPlugin): The object of the plugin that will be executed. + command (HermesCommand): The command object containing information on the run. + + Returns: + ld_dict: The provenance entity of the plugin (can be used to update the data in the ld_prov_list). + """ + # construct basic data dict + data = { + "@id": ld_prov_list.HERMES_PLUGIN_ID_FORMAT.format(step=step, name=name), + "@type": "schema:SoftwareApplication", + "schema:name": f"{plugin.__module__}.{plugin.__class__.__qualname__}", + "schema:description": f"{utils.hermes_name} {step} plugin '{name}'", + "schema:supportingData": { + "@type": "schema:DataFeed", + "schema:dataFeedElement": [] + }, + "prov:actedOnBehalfOf": self.get_hermes_base_plugin(step).ref + } + # try adding the settings for the plugin + try: + for name, values in getattr(command.settings, name).model_dump(mode="json").items(): + if not isinstance(values, list): + values = [values] + data["schema:supportingData"]["schema:dataFeedElement"].append({ + "@type": "schema:DataFeedItem", + "schema:name": name, + "schema:item": [ + { + "@type": "schema:Item", + "schema:description": value + } + for value in values + ] + }) + except Exception: + del data["schema:supportingData"] + # try adding the version of the package of the plugin + try: + data["schema:softwareVersion"] = metadata(plugin.__module__.split(".")[0])["version"] + except Exception: + pass + # add the plugin to the ld_prov_list and return the object + node = self.add_agent(data=data) + return node + + def shallow_search(self: Self, query: Callable[[ld_dict], Any]) -> list[ld_dict]: + """ + Search the objects in the ld_prov_list for objects for which the query evaluates to True. + + Args: + query (Callable[[ld_dict], Any]): The query used for evaluating the objects. + + Returns: + list[ld_dict]: The objects in the ld_prov_list for which `query` evalutes to True. + """ + return [item for item in self if query(item)] + + def get_hermes(self: Self) -> ld_dict: + """ + Returns the hermes agent in the ld_prov_list. + + Returns: + ld_dict: The object representing the hermes agent. + """ + return self.shallow_search(lambda node: ("@id" in node and node["@id"] == ld_prov_list.HERMES_ID))[0] + + def get_hermes_cache(self: Self) -> ld_dict: + """ + Returns the hermes cache agent in the ld_prov_list. + + Returns: + ld_dict: The object representing the hermes cache agent. + """ + return self.shallow_search(lambda node: ("@id" in node and node["@id"] == ld_prov_list.HERMES_CACHE_ID))[0] + + def get_hermes_base_plugin(self: Self, step: str) -> ld_dict: + """ + Returns the base plugin agent in the ld_prov_list of the given step. + + Args: + step (str): The step of which the base plugin agent should be returned. + + Returns: + ld_dict: The object representing the base plugin agent of the given step. + """ + return self.shallow_search(lambda node: ( + "@id" in node and node["@id"] == ld_prov_list.HERMES_BASE_PLUGIN_ID_FORMAT.format(step=step) + ))[0] + + def get_hermes_plugin(self: Self, step: str, name: str) -> Union[ld_dict, None]: + """ + Returns the plugin agent in the ld_prov_list of the given step with the given name. + + Args: + step (str): The step of which the plugin agent should be returned. + name (str): The name of the plugin agent that should be returned. + + Returns: + ld_dict | None: The object representing the plugin agent of the given step with the given name. + """ + search_result = self.shallow_search(lambda node: ( + "@id" in node and node["@id"] == ld_prov_list.HERMES_PLUGIN_ID_FORMAT.format(step=step, name=name) + )) + if search_result: + return search_result[0] + return None + + def get_hermes_command(self, step) -> ld_dict: + """ + Returns the hermes command agent in the ld_prov_list of the given step. + + Args: + step (str): The step of which the hermes command agent should be returned. + + Returns: + ld_dict: The object representing the hermes command agent of the given step. + """ + return self.shallow_search(lambda node: ( + "@id" in node and node["@id"] == ld_prov_list.HERMES_COMMAND_ID_FORMAT.format(step=step) + ))[0] diff --git a/src/hermes/model/types/ld_container.py b/src/hermes/model/types/ld_container.py index ea4c8dbd..83337072 100644 --- a/src/hermes/model/types/ld_container.py +++ b/src/hermes/model/types/ld_container.py @@ -13,6 +13,8 @@ from typing import Any, Optional, TypeAlias, TYPE_CHECKING, Union from typing_extensions import Self +from hermes.model.types.ld_context import iri_map + from .pyld_util import JsonLdProcessor, bundled_loader if TYPE_CHECKING: from .ld_dict import ld_dict @@ -231,7 +233,7 @@ def _to_expanded_json( # all ld_container (ld_dicts and ld_lists) and datetime, date as well as time objects in value have to dissolved # because the JSON-LD processor can't handle them # to do this traverse value in a BFS and replace all items with a type in 'special_types' with a usable values - key_and_reference_todo_list = [(0, [value])] + key_and_reference_todo_list: list[Union[tuple[int, list], tuple[str, dict]]] = [(0, [value])] special_types = (list, dict, ld_container, datetime, date, time) while True: # check if ready @@ -317,7 +319,7 @@ def compact( COMPACTED_JSON_LD_VALUE: The compacted version of selfs JSON-LD representation. """ return self.ld_proc.compact( - self.ld_value, context or self.context, {"documentLoader": bundled_loader, "skipExpand": True} + self.ld_value, context or self.full_context, {"documentLoader": bundled_loader, "skipExpand": True} ) def to_native_python(self): @@ -461,8 +463,13 @@ def typed_ld_to_py(cls: type[Self], data: list[dict[str, BASIC_TYPE]], **kwargs) Returns: BASIC_TYPE | TIME_TYPE: The native python version of data. """ - # FIXME: #434 dates are not returned as datetime/ date/ time but as string ld_value = data[0]['@value'] + if iri_map["schema:DateTime"] == data[0]['@type']: + ld_value = datetime.fromisoformat(ld_value) + elif iri_map["schema:Date"] == data[0]['@type']: + ld_value = date.fromisoformat(ld_value) + elif iri_map["schema:Time"] == data[0]['@type']: + ld_value = time.fromisoformat(ld_value) return ld_value diff --git a/src/hermes/model/types/ld_context.py b/src/hermes/model/types/ld_context.py index 712c8bd8..ecf53dd8 100644 --- a/src/hermes/model/types/ld_context.py +++ b/src/hermes/model/types/ld_context.py @@ -128,7 +128,7 @@ def __getitem__(self: Self, compressed_term: Union[str, tuple]) -> str: Raises: HermesCacheError: If the compressed term is '' or its prefix can't be expanded. """ - # seperate the prefix from the term + # separate the prefix from the term if not isinstance(compressed_term, str): prefix, term = compressed_term elif ":" in compressed_term: diff --git a/src/hermes/model/types/ld_dict.py b/src/hermes/model/types/ld_dict.py index faaffe57..f6ec1c6b 100644 --- a/src/hermes/model/types/ld_dict.py +++ b/src/hermes/model/types/ld_dict.py @@ -8,6 +8,7 @@ from __future__ import annotations from collections.abc import Generator, Iterator, KeysView +from datetime import date, datetime, time from typing import Any, Literal, Optional, Union, TYPE_CHECKING from typing_extensions import Self @@ -383,6 +384,39 @@ def from_dict( Returns: ld_dict: The new ld_dict build from value. """ + # all ld_container (ld_dicts and ld_lists) and datetime, date as well as time objects in value have to dissolved + # because the JSON-LD processor can't handle them + # to do this traverse value in a BFS and replace all items with a type in 'special_types' with a usable values + key_and_reference_todo_list: list[Union[tuple[int, list], tuple[str, dict]]] = [(0, [value])] + special_types = (list, dict, ld_container, datetime, date, time) + while True: + # check if ready + if len(key_and_reference_todo_list) == 0: + break + # get next item + tmp_key, ref = key_and_reference_todo_list.pop() + temp = ref[tmp_key] + # replace item if necessary and add childs to the todo list + if isinstance(temp, list): + key_and_reference_todo_list.extend( + [(index, temp) for index, val in enumerate(temp) if isinstance(val, special_types)] + ) + elif isinstance(temp, dict): + key_and_reference_todo_list.extend( + [(new_key, temp) for new_key in temp.keys() if isinstance(temp[new_key], special_types)] + ) + elif isinstance(temp, ld_container): + if "ld_list" in [sub_cls.__name__ for sub_cls in type(temp).mro()] and temp.container_type == "@set": + ref[tmp_key] = temp._data + else: + ref[tmp_key] = temp._data[0] + elif isinstance(temp, datetime): + ref[tmp_key] = {"@value": temp.isoformat(), "@type": "schema:DateTime"} + elif isinstance(temp, date): + ref[tmp_key] = {"@value": temp.isoformat(), "@type": "schema:Date"} + elif isinstance(temp, time): + ref[tmp_key] = {"@value": temp.isoformat(), "@type": "schema:Time"} + # make a copy of value and add the new type to it. ld_data = value.copy() ld_type = ld_container.merge_to_list(ld_type or [], ld_data.get('@type', [])) diff --git a/src/hermes/model/types/ld_list.py b/src/hermes/model/types/ld_list.py index d8bfcf0f..4d6aa923 100644 --- a/src/hermes/model/types/ld_list.py +++ b/src/hermes/model/types/ld_list.py @@ -15,6 +15,7 @@ from typing_extensions import Self from .ld_container import ( + COMPACTED_JSON_LD_VALUE, ld_container, JSON_LD_CONTEXT_DICT, EXPANDED_JSON_LD_VALUE, @@ -23,6 +24,7 @@ TIME_TYPE, BASIC_TYPE, ) +from .pyld_util import bundled_loader if TYPE_CHECKING: from .ld_dict import ld_dict @@ -548,6 +550,45 @@ def to_native_python(self: Self) -> list[Union[BASIC_TYPE, TIME_TYPE, NATIVE_LD_ for item in self ] + def compact( + self: Self, context: Optional[Union[list[Union[JSON_LD_CONTEXT_DICT, str]], JSON_LD_CONTEXT_DICT, str]] = None + ) -> COMPACTED_JSON_LD_VALUE: + """ + Returns the compacted version of the given ld_list using its context only if none was supplied. + The returned object is of the form `{"@context": the_context, container_type: compacted_content}`. + + Args: + context (list[JSON_LD_CONTEXT_DICT | str] | JSON_LD_CONTEXT_DICT | str | None): + The context to use for the compaction. If None the context of self is used. + + Returns: + COMPACTED_JSON_LD_VALUE: The compacted version of selfs JSON-LD representation. + """ + # compact the ld_list standalone if necessary + if self.key is None: + return self.ld_proc.compact( + self.ld_value, context or self.full_context, {"documentLoader": bundled_loader, "skipExpand": True} + ) + # compact the ld_list within a temporary dictionary + temp_dict = self.ld_proc.compact( + [{self.ld_proc.expand_iri(self.active_ctx, self.key): self.ld_value}], + context or self.full_context, + {"documentLoader": bundled_loader, "skipExpand": True} + ) + context = temp_dict["@context"] + temp_container = temp_dict[ + self.ld_proc.compact_iri(self.active_ctx, self.ld_proc.expand_iri(self.active_ctx, self.key)) + ] + if self.container_type != "@set": + return { + "@context": context, + **temp_container + } + return { + "@context": context, + "@set": temp_container if isinstance(temp_container, list) else [temp_container] + } + @classmethod def is_ld_list(cls: type[Self], ld_value: Any) -> bool: """ @@ -589,7 +630,7 @@ def from_list( key: Optional[str] = None, context: Optional[Union[str, JSON_LD_CONTEXT_DICT, list[Union[str, JSON_LD_CONTEXT_DICT]]]] = None, container_type: str = "@set" - ) -> ld_list: + ) -> Self: """ Creates a ld_list from the given list with the given parent, key, context and container_type.\n Note that only container_type '@set' is valid for key '@type'.\n @@ -614,7 +655,6 @@ def from_list( Raises: ValueError: If key is '@type' and container_type is not '@set'. """ - # TODO: handle context if not of type list or None # validate container_type if key == "@type": if container_type != "@set": @@ -626,6 +666,10 @@ def from_list( elif container_type != "@set": raise ValueError(f"Invalid container type: {container_type}. (valid are only '@set', '@list' and '@graph')") + # handle non-list context + if context is not None and not isinstance(context, list): + context = [context] + if parent is not None: # expand value in the "context" of parent if isinstance(parent, ld_list): diff --git a/test/hermes_test/commands/postprocess/test_invenio_postprocess.py b/test/hermes_test/commands/postprocess/test_invenio_postprocess.py index 3a2a77fc..7150d15f 100644 --- a/test/hermes_test/commands/postprocess/test_invenio_postprocess.py +++ b/test/hermes_test/commands/postprocess/test_invenio_postprocess.py @@ -58,7 +58,7 @@ def test_invenio_postprocess(tmp_path, monkeypatch): communities = "api/communities" [postprocess] -run = ["config_invenio_record_id", "cff_doi", "codemeta_doi"] +run = ["config_invenio_record_id", "invenio_cff_doi", "invenio_codemeta_doi"] """ ) @@ -101,7 +101,7 @@ def test_invenio_postprocess(tmp_path, monkeypatch): communities = "api/communities" [postprocess] -run = ["config_invenio_record_id", "cff_doi", "codemeta_doi"] +run = ["config_invenio_record_id", "invenio_cff_doi", "invenio_codemeta_doi"] """ ).unwrap() assert result_cff == yaml.YAML().load( diff --git a/test/hermes_test/model/test_api.py b/test/hermes_test/model/test_api.py index 68cb73cf..f01d5370 100644 --- a/test/hermes_test/model/test_api.py +++ b/test/hermes_test/model/test_api.py @@ -143,6 +143,5 @@ def test_usage(): if "Baz" not in author["name"]: assert "email" in author if "schema:knowsAbout" not in author: - # FIXME: None has to be discussed author["schema:knowsAbout"] = None author["schema:pronouns"] = "they/them" diff --git a/test/hermes_test/model/types/test_ld_container.py b/test/hermes_test/model/types/test_ld_container.py index 9cf8f871..0442aef6 100644 --- a/test/hermes_test/model/types/test_ld_container.py +++ b/test/hermes_test/model/types/test_ld_container.py @@ -122,10 +122,12 @@ def test_to_native_python_basic_value(self, mock_context): def test_to_native_python_datetime_value(self, mock_context): cont = ld_container([{}], context=[mock_context]) - assert cont._to_native_python( + res = cont._to_native_python( "http://example.com/eggs", - {"@value": "2022-02-22T00:00:00", "@type": "https://schema.org/DateTime"} - ) == "2022-02-22T00:00:00" # TODO: #434 typed date is returned as string instead of date + {"@value": "2022-02-22T00:00:00", "@type": "http://schema.org/DateTime"} + ) + assert isinstance(res, datetime) + assert res == datetime.fromisoformat("2022-02-22T00:00:00") def test_to_native_python_error(self, mock_context): cont = ld_container([{}], context=[mock_context]) diff --git a/test/hermes_test/model/types/test_ld_dict.py b/test/hermes_test/model/types/test_ld_dict.py index ca2a7376..faba0acf 100644 --- a/test/hermes_test/model/types/test_ld_dict.py +++ b/test/hermes_test/model/types/test_ld_dict.py @@ -5,6 +5,8 @@ # SPDX-FileContributor: Stephan Druskat # SPDX-FileContributor: Michael Fritzsche +from datetime import datetime + import pytest from hermes.model.types.ld_dict import ld_dict @@ -407,6 +409,39 @@ def test_from_dict(): assert di["http://xmlns.com/foaf/0.1/name"] == di["xmlns:name"] == ["fo"] assert di.context == [{"schema": "https://schema.org/"}, {"xmlns": "http://xmlns.com/foaf/0.1/"}] + di = ld_dict.from_dict( + { + "@context": {"schema": "http://schema.org/"}, + "@type": "schema:Thing", + "schema:owner": ld_dict.from_dict( + { + "@context": {"schema": "http://schema.org/"}, + "@type": "schema:Person", + "schema:name": "Foo" + } + ) + } + ) + assert di["schema:owner"][0]["schema:name"][0] == "Foo" + di = ld_dict.from_dict( + { + "@context": {"schema": "http://schema.org/"}, + "@type": "schema:Thing", + "schema:name": ld_list.from_list( + ["Foo", "Bar"], key="schema:name", context={"schema": "http://schema.org/"}, container_type="@list" + ) + } + ) + assert di["schema:name"][0] == "Foo" + di = ld_dict.from_dict( + { + "@context": {"schema": "http://schema.org/"}, + "@type": "schema:CreativeWork", + "schema:dateCreated": datetime(2026, 8, 31, 14, 50) + } + ) + assert di["schema:dateCreated"][0] == datetime(2026, 8, 31, 14, 50) + def test_is_ld_dict(): assert not any(ld_dict.is_ld_dict(item) for item in [{}, {"foo": "bar"}, {"@id": "foo"}])