diff --git a/.forgejo/workflows/scrape-juser.yml b/.forgejo/workflows/scrape-juser.yml new file mode 100644 index 0000000..b2f24fe --- /dev/null +++ b/.forgejo/workflows/scrape-juser.yml @@ -0,0 +1,47 @@ +name: Scrape publications from JUSER + +on: + workflow_dispatch: + inputs: + year: + description: "Fetch Juser publications from this year (e.g. 2026)" + required: true + default: '2026' + type: string + +env: + DTC_TOKEN: ${{ secrets.POOLTOKEN }} + DUMPTHINGS_APIURL: https://pool.psychoinformatics.de/api + DUMPTHINGS_COLLECTION: public + JUSER_PW: ${{ secrets.JUSER_PW }} + JUSER_USER: ${{ secrets.JUSER_USER }} + +jobs: + scrape-juser: + name: Scrape JUSER + runs-on: debian-latest + defaults: + run: + shell: bash + steps: + - name: Install uv + uses: astral-sh/setup-uv@v6 + - name: Install metadata tools + run: | + uv tool install https://hub.psychoinformatics.de/orinoco/query-things.git \ + --with-executables-from dump-things-pyclient + - name: Fetch script + run: | + wget https://hub.psychoinformatics.de/orinoco/knowledge-enrichment/raw/branch/main/tools/scrape-juser.py + - name: Pre-fetch pool data + run: | + mkdir .cache + dtc get-records $DUMPTHINGS_APIURL public -C XYZPublication > .cache/Publications.jsonl + dtc get-records $DUMPTHINGS_APIURL public -C XZYPerson > .cache/Person.jsonl + dtc get-records $DUMPTHINGS_APIURL protected -C XZYPerson >> .cache/Person.jsonl + - name: Pre-fetch Juser records + curl 'https://juser.fz-juelich.de/PubExporter.py?p=cid%3A%22I%3A%28DE-Juel1%29INM-7-20090406%22+AND+pub%3A%22${{ inputs.year }}%22&sf=author&so=d&rg=&of=xm' > .cache/juser-pubs.xml + + - name: Process records + run: | + uv run scrape-juser.py --file .cache/juser-pubs.xml --persons .cache/Person.jsonl --publications .cache/Publications.jsonl diff --git a/README.md b/README.md index 706ab10..5dc3ffa 100644 --- a/README.md +++ b/README.md @@ -11,4 +11,5 @@ with additional machine-generated records. - ``tools/enrich-via-doi.py`` (used by ``.forgejo/workflows/enrich_publications.yml``): reads publication records and extends them with external metadata available via doi.org content negotiation -- ``tools/get-depiction-urls.py`` (used by the 'from-model'-websites, e.g., https://hub.psychoinformatics.de/www/www-from-model/src/branch/main/.forgejo/workflows/register-depictions.yaml): Given metadata input, this script extracts download URLs for each ``Depiction`` of the record, provided the depiction has a ``kind`` included in the depiction-type argument. For each depiction distribution, it will output the record curie, the file extension and the url, to stdout. \ No newline at end of file +- ``tools/get-depiction-urls.py`` (used by the 'from-model'-websites, e.g., https://hub.psychoinformatics.de/www/www-from-model/src/branch/main/.forgejo/workflows/register-depictions.yaml): Given metadata input, this script extracts download URLs for each ``Depiction`` of the record, provided the depiction has a ``kind`` included in the depiction-type argument. For each depiction distribution, it will output the record curie, the file extension and the url, to stdout. +- ``tools/scrape-juser.py`` (used by ``.forgejo/workflows/scrape-juser.yml``): Given an xml export of publications in MARC21 format as provided by Juser (juser.fz-juelich.de), as well as a local cache of XYZPerson and XYZPublication records, this script generates new or updates existing Publication and Person records. Person records are submitted into a 'protected' inbox, whereas Publication records are submitted into a 'public' inbox. \ No newline at end of file diff --git a/tools/scrape-juser.py b/tools/scrape-juser.py new file mode 100644 index 0000000..b6a6b0c --- /dev/null +++ b/tools/scrape-juser.py @@ -0,0 +1,841 @@ +# /// script +# requires-python = ">=3.12" +# dependencies = [ +# "dump-things-pyclient @ https://hub.psychoinformatics.de/datalink/dump-things-pyclient.git", +# "pymarc", +# "rich", +# "rich-click", +# ] +# /// + +""" +Using a local xml file with MARCXML publication data from JUSER, this script +creates and submits publication records to a given pool. + +Usage: +1) Download xml data with one of JUSERs generated search URLs +(see https://juser.fz-juelich.de/search_generator.py) +> mkdir .cache +> curl 'https://juser.fz-juelich.de/PubExporter.py?p=cid%3A%22I%3A%28DE-Juel1%29INM-7-20090406%22+AND+pub%3A%222026%22&sf=author&so=d&rg=&of=xm' > .cache/juser-pubs.xml + +2) Cache Person and Publication records from your pool +> dtc get-records $DUMPTHINGS_APIURL \ + public -C XYZPublication > .cache/Publications.jsonl +> dtc get-records $DUMPTHINGS_APIURL \ + public -C XYZPerson > .cache/Person.jsonl + +3) Invoke the script +> uv run tools/scrape-juser.py \ + --file .cache/juser-pubs.xml \ + --persons /tmp/.cache/Person.jsonl \ + --publications /tmp/.cache/Publications.jsonl + +""" +import click +import json +import logging +import requests +import uuid +from collections.abc import Callable +from pymarc import parse_xml_to_array +from os import environ +from pathlib import Path +from urllib import parse +from requests import Session +from xml.etree import ElementTree as ET + +from dump_things_pyclient.communicate import collection_write_record + + +def juser_session( + url: str = 'https://juser.fz-juelich.de/youraccount/login', +): + """Create a Session object in order to have a persisting session token + for authentication. This is required for Person searches.""" + data = {'login_method': 'FZJ eMail-Account', + 'p_un': environ.get('JUSER_USER'), + 'p_pw': environ.get('JUSER_PW'), + 'action': 'login'} + s = Session() + # required to solve challenge + s.cookies.set('APP_INIT', '1') + response = s.post(url, + data=data) + assert response.status_code == 200 + return s + + +def _lookup( + record, + field1: str, + field2: str +) -> str | None: + """Helper function to look up metadata without running into KeyErrors""" + parentfield = record.get(field1, None) + if parentfield is not None: + value = parentfield.get(field2, None) + if value is not None: + return value + return None + + +class JuserScraper(object): + def __init__( + self, + pool: str, + collection: str, + xml: str, + persons, + pubs + ) -> None: + self.pool = pool + self.collection = collection + self.xml = Path(xml) + # caches of dumpthings records + self.persons = map_record_to_feature(process_orcid, persons) + self.pubs = map_record_to_feature(process_doi, pubs) + # cache of JulID to ORCID associations + self.JulIDs = {} + # this list stores to-be-submitted records + self.to_submit_public = [] + self.to_submit_protected = [] + # establish a session to Juser + self.session = juser_session() + # potentially make this a parameter + self.scriptpid = 'xyzrins:instruments/d3126ff5-623c-48af-ac48-4d921ef9b80d' + + def create_records( + self + ) -> None: + + records = parse_xml_to_array(self.xml) + logging.info('Found {} publication records'.format(len(records))) + + # Loop over all records + for r in records: + self.retrieve_metadata(r) + + def submit_records( + self, + records: list, + collection: str, + class_name: str, + ) -> None: + # finally, submit: + for record in records: + print(f"submitting record with pid {record['pid']}" + f" to collection {collection}") + try: + collection_write_record( + service_url=self.pool, + collection=collection, + class_name=class_name, + record=record, + format='json', + token=environ['DTC_TOKEN'] + ) + except requests.exceptions.HTTPError: + print("SUBMISSION ERROR FOR RECORD: ") + print(json.dumps(record)) + return + + def _check_if_mutable( + self, + record: dict, + k: str, + only_self_edits: bool = False, + predicate: str | None = None, + ) -> bool: + """If a record either does not already have the info, or the info is + annotated to be machine-generated, allow overwriting it. + params: + record: dict -> the metadata record + k: str -> record key to check + only_self_edits: True|False -> only allows overwriting existing infos if + they stem from the same script + """ + + if k not in record.keys(): + return True + infos = record[k] + if type(infos) == dict and 'annotations' in infos.keys(): + importedBy = \ + infos.get('annotations', {}).get('http://purl.org/pav/importedBy', 'nothing') + return self._is_machine_generated(importedBy, only_self_edits) + # if the key is a data property, infos is just a string. + attributes = record.get('attributes', [{}]) + for attribute in attributes: + if attribute.get('predicate', None) == predicate: + return \ + self._is_machine_generated(attribute.get('annotations', {}).get('http://purl.org/pav/importedBy', 'nothing')) + + def _is_machine_generated( + self, + importedBy, + only_self_edits: bool = False, + ): + if importedBy.startswith('xyzrins:instruments'): + if not only_self_edits: + return True + else: + return importedBy == self.scriptpid + return False + + def _get_data_property( + self, + record: dict, + prop: str, + predicate: str, + lookup: tuple + ): + if self._check_if_mutable(record, prop, predicate=predicate): + return _lookup(lookup[0], lookup[1], lookup[2]) + return None + + def _get_attribution( + self, + record, + r, + annotations + ) -> dict: + authors = r.get_fields('700') + new_attributions = [] + for author in authors: + found = False + orcid = None + JulID = None + # if the author haus an FZJ affiliation, get their info. We can most + # reliably look up Orcid based on Julich ID -- external authors + # often are only mentioned by name + if '(DE-Juel1)' in author.subfields[1].value: + JulID = author.subfields[1].value + display_label = author.subfields[0].value + orcid, record_id = self._look_up_orcid(JulID) + logging.info(f'Found author {display_label} under ID {JulID} ' + f'with orcid {orcid}: ' + f'https://juser.fz-juelich.de/record/{record_id}') + mainid = orcid + if orcid is None: + if not JulID: + # stop the author has no orcid and their JulID is also not + # known + continue + # we can make a second attempt with a Julich-issued ID + mainid = self.persons.get(JulID, None) + if not mainid: + continue + # either get the pid of the preexisting record or build new pid + pid = self.persons.get(mainid, {}).get('pid', + 'xyzrins:persons/' + str(uuid.uuid4())) + person_annotations = \ + {'http://purl.org/pav/importedBy': self.scriptpid, + 'http://purl.org/pav/importedFrom': + f'https://juser.fz-juelich.de/record/{record_id}'} + # create the author attribution: + attr = {"schema_type": "dlthings:Attribution", + "object": pid, + "annotations": annotations} + # submit new person record, if not yet existing + if mainid not in self.persons.keys(): + self._new_person( + p_pid=pid, + display_label=display_label, + orcid=orcid, + JulID=JulID, + annotations=person_annotations + ) + # update the person record, if Juser had any additional infos + else: + self._update_person( + mainid=mainid, + orcid=orcid, + JulID=JulID, + annotations=person_annotations + ) + # check if the author is already attributed in the record + for i, attribution in enumerate(record.get('attributed_to', [{}])): + if pid == attribution.get('object'): + # the author is known. Check if editable, if so, add author + # and annotation to the record: + if self._is_machine_generated( + attribution.get('annotations', {}).get('http://purl.org/pav/importedBy', 'nothing')): + if attribution != attr: + # only update if the record differs + record['attributed_to'][i] = attr + # search is over once we found the author, continue next + found = True + continue + if not found: + new_attributions.append(attr) + old_attributions = record.get('attributed_to', []) + old_attributions.extend(new_attributions) + return old_attributions + + def _new_person( + self, + p_pid, + display_label, + orcid, + JulID, + annotations + ): + print('CREATING NEW PERSON RECORD FOR ', display_label) + p_rec = {'pid': p_pid, + # MARCXML does only report author names as one string + # with varying formatting (incl. or excl. initials, + # etc). Parsing this into first name and family name + # is error-prone. Instead, we use their string as a + # display label + 'display_label': display_label, + 'identifiers': [ + {'schema_type': 'xyzri:ORCID', + 'creator': 'ror:04fa4r544', + 'notation': orcid, + 'annotations': annotations}, + {'schema_type': 'dlthings:Identifier', + 'creator': 'https://w3id.org/isil/DE-Juel1', + # ZB FZJ + 'notation': JulID, + 'annotations': annotations}], + } + # collect to be created record for submission. Person records + # are protected + self.to_submit_protected.append(p_rec) + # store the author in internal cache to not resubmit + self.persons[orcid] = p_rec + return + + def _get_generation( + self, + record, + r, + annotations + ): + generated_by = record.get('generated_by', []) + new_generations = [] + # extract which pof project generated a publication. This may + # be several! If no PoF is found, gen will be an empty list + pof = self._get_pof(r, annotations) + if pof: + for p in pof: + # is there an existing record, and can we edit it? + pof_pid = p['object'] + found = False + for i, generation in enumerate(generated_by): + found = False + if generation.get('object', None) == pof_pid: + found = True + if self._is_machine_generated( + generation.get('annotations', {}).get('http://purl.org/pav/importedBy', 'nothing')): + if p != generation: + # overwrite + generated_by[i] = p + if not found: + new_generations.append(p) + # next, record the publication process + date = _lookup(r, '773', 'y') + at_location = self._get_periodical(r) + if date is not None and at_location is not None: + publication_process = \ + {"at_location": "ISSN:{}".format(at_location), + "at_time": "{}".format(date), + "schema_type": "dlthings:Generation", + "object": "obo:IAO_0000444", + "annotations": annotations} + # is there an existing record, and can we edit it? + periodical_pid = publication_process['object'] + for i, generation in enumerate(generated_by): + found = False + if generation.get('object', 'nothing') == periodical_pid: + if self._is_machine_generated( + generation.get('annotations', {}).get('http://purl.org/pav/importedBy', 'nothing')): + if publication_process != generation: + # overwrite + generated_by[i] = publication_process + found = True + if not found: + new_generations.append(publication_process) + generated_by.extend(new_generations) + return generated_by + + def _get_influence( + self, + record, + r, + annotations + ): + influenced_by = record.get('influenced_by', []) + new_influences = [] + funding = self._get_funding(r, annotations) + if funding: + for fund in funding: + # check if the funding is already part of the record + fund_pid = fund.get('object', None) + for i, influence in enumerate(influenced_by): + found = False + if influence.get('object', 'nothing') == fund_pid: + if self._is_machine_generated( + influence.get('annotations', {}).get('http://purl.org/pav/importedBy', None)): + if fund != influence: + # overwrite + influenced_by[i] = fund + found = True + if not found: + new_influences.append(fund) + influenced_by.extend(new_influences) + return influenced_by + + def _get_pof( + self, + r, + annotations + ) -> list | None: + all_funding = r.get_fields('536') + # for XYZPublication in Research Information, generation is multivalued + pof_gen = [] + for funding in all_funding: + fund = funding['0'] + # this fund may be a PoF association or a grant + pof_pid = map_pof_to_pid.get(fund, None) + if pof_pid is not None: + pof_gen.append({"schema_type": "dlthings:Generation", + "object": pof_pid, + "annotations": annotations}) + return pof_gen + + def _get_funding( + self, + r, + annotations, + ) -> list | None: + all_funding = r.get_fields('536') + funding_ack = [] + for funding in all_funding: + fund = funding['0'] + # this fund may be a PoF association or a grant + fund_rec = map_grant_to_pid.get(fund, None) + if fund_rec is not None: + funding_ack.append( + {"schema_type": "xyzri:XYZInfluence", + "object": fund_rec['pid'], + "roles": ["FRAPO:Funding"], + "ended": {"at_time": fund_rec['end'], + "schema_type": "dlthings:End"}, + "started": {"at_time": fund_rec['start'], + "schema_type": "dlthings:Start"}, + "annotations": annotations}) + return funding_ack + + def _update_person( + self, + mainid, + orcid, + JulID, + annotations, + ): + """for an existing person record, check ORCID or JulID can be added""" + metadata_JulID = {"schema_type": "dlthings:Identifier", + "creator": "https://w3id.org/isil/DE-Juel1", + "notation": JulID, + "annotations": annotations} + metadata_orcid = {"schema_type": "xyzri:ORCID", + "creator": "ror:04fa4r544", + "notation": orcid, + "annotations": annotations} + person = self.persons.get(mainid) + changes = False + for ident, lookupkey, lookupvalue, meta in \ + [(orcid, 'schema_type', 'xyzri:ORCID', metadata_orcid), + (JulID, 'creator', 'https://w3id.org/isil/DE-Juel1', metadata_JulID)]: + found = False + if ident is None: + continue + for identifier in person.get("identifiers", []): + # what kind of identifier is it? + if identifier.get(lookupkey, None) == lookupvalue: + # found the identifier, no need to add it + found = True + if not found: + print(f"UPDATING PERSON RECORD WITH ID {mainid} WITH {ident}") + person["identifiers"].append(meta) + changes = True + if changes: + # only update record if a change was made + self.persons[mainid] = person + self.to_submit_protected.append(person) + return + + def retrieve_metadata( + self, + r, + ): + """For a given MARCXML publication record from Juser, check if a + publication with this DOI is already in the pool. If a record already + exists, check if it needs amendments. If a record does not exist, + assemble a publication metadata record from scratch""" + # first, get the DOI and check if the publication already exists + doi = self._get_doi(r) + if doi is None: + return + # record of the juser ID, for machine provenance and linking to the + # original + juser_id = r.fields[0].data + annotations = {"http://purl.org/pav/importedBy": self.scriptpid, + "http://purl.org/pav/importedFrom": f'https://juser.fz-juelich.de/record/{juser_id}'} + if doi in self.pubs.keys(): + oldrecord = self.pubs[doi] + record = oldrecord + print('CHECKING EXISTING PUBLICATION WITH DOI ', doi) + # because JUSER is stupid: + elif doi.lower() in self.pubs.keys(): + oldrecord = self.pubs[doi.lower()] + record = oldrecord + print('CHECKING EXISTING PUBLICATION WITH DOI ', doi) + else: + print('CREATING NEW PUBLICATION FOR DOI ', doi) + oldrecord = None + record = {'schema_type': 'xyzri:XYZPublication', + 'pid': 'xyzrins:publications/' + str(uuid.uuid4()), + 'identifiers': [ + {'schema_type': 'dlthings:DOI', + 'notation': doi, + 'annotations': annotations} + ] + } + # the next steps obtain metadata fields IF the record allows overwriting + title = \ + self._get_data_property(record, 'title', 'dcterms:title', + (r, '245', 'a')) + description = \ + self._get_data_property(record, 'description', 'dcterms:abstract', + (r, '520', 'a')) + kind = \ + self._get_data_property(record, 'kind', 'dcterms:type', + (r, '336', 'a')) + content_type = map_content_to_bibitem.get(kind, None) \ + if kind is not None else None + kind = content_type + # add empty attributes for machine prov for data properties + # TODO: the slot could also be determined by a different script + attributes = record.get('attributes', []) + for prop, slot, term in [(title, 'title', 'dcterms:title'), + (description, 'description', 'dcterms:abstract'), + (kind, 'kind', 'dcterms:type')]: + if prop is not None: + record[slot] = prop + metadata = {'predicate': term, + 'value': prop, + 'annotations': annotations} + if attributes: + for i, attr in enumerate(attributes): + found = False + if attr.get('predicate', None) == term: + attributes[i] = metadata + if not found: + attributes.append(metadata) + else: + attributes.append(metadata) + record['attributes'] = attributes + + attributed_to = self._get_attribution(record, r, annotations) + generated_by = self._get_generation(record, r, annotations) + influenced_by = self._get_influence(record, r, annotations) + for prop, slot in [(attributed_to, 'attributed_to'), + (generated_by, 'generated_by'), + (influenced_by, 'influenced_by')]: + if prop: + record[slot] = prop + if oldrecord != record: + # publication records can be public + print('FOUND AN UPDATE FOR ', doi) + self.to_submit_public.append(record) + return + + def _get_doi( + self, + r + ) -> str | None: + doi = None + # check if field 024 exists + test = _lookup(r, '024', 'a') + if test: + if r['024'].subfields[1].value == 'doi': + doi = r['024']['a'] + return doi + + def _get_periodical( + self, + r + ) -> str | None: + """Obtain the ISSN of a publication venue""" + # The ZB identifies periodicals via an internal ID, e.g., + # 'PERI:(DE-600)2808093-2'We can look up the ISSN + # for an ID in their periodicals collection: + # https://juser.fz-juelich.de/collection/Periodicals?ln=en + peri = _lookup(r, '773', '0') + if peri is None: + return None + ID = parse.quote_plus(peri) + url = \ + f'https://juser.fz-juelich.de/search?ln=en&cc=Periodicals&p={ID}&f=&action_search=Search&c=Periodicals&c=&sf=&so=d&rm=&rg=10&sc=1&of=xm' + x = self._look_up_juser_info(url, attempts=3) + # look up ISSN in the XML above. XML is a mess. + try: + ISSN = x.findall(".//*[@tag='022']")[0][0].text + except IndexError as e: + ISSN = None + return ISSN + + def _look_up_juser_info( + self, + url: str, + attempts: int = 3 + ): + # The retrieval of search results often glitches when ran in short + # succession, so we safe-guard and retry + lookup = False + attempt = 0 + while not lookup and attempt < attempts: + try: + res = self.session.get(url) + attempt += 1 + x = ET.fromstring(res.content.decode()) + except ET.ParseError as e: + print("JUSER GLITCH, RETRYING...") + continue + lookup = True + return x + + def _look_up_orcid( + self, + julid: str, + ) -> (str, str): + if julid in self.JulIDs.keys(): + # this ID is already known, don't query again + return self.JulIDs[julid], None + # URL encode the ID + ID = parse.quote_plus(julid) + url = \ + f'https://juser.fz-juelich.de/search?ln=en&p={ID}&f=&action_search=Search&c=People&sf=author&so=d&rm=&rg=10&sc=0&of=xm' + x = self. _look_up_juser_info(url, attempts=3) + # look up ORCID in the XML above. XML is a mess. + try: + ORCID = \ + dict((e[0].text.casefold(), e[1].text) for e in + x.findall(".//*[@tag='024']"))[ + 'orcid'] + # add to cache + self.JulIDs[julid] = ORCID + except KeyError as e: + ORCID = None + try: + record_id = [e.text for e in x.findall(".//*[@tag='001']")][0] + record_source = f'https://juser.fz-juelich.de/record/{record_id}' + except Exception as e: + record_source = None + return ORCID, record_source + + +def map_record_to_feature( + feature: Callable, + records: list[dict] +) -> dict[str, str]: + """Create a mapping of records to a specific feature (DOI or ORCIDs) + """ + my_map = {} + for record in records: + if (feat := feature(record)) is not None: + my_map[feat] = record + return my_map + + +# TODO: taken from enrich-via-doi.py -- factor out and import! +def process_doi(paper: dict) -> str | None: + """Return a DOI from identifiers""" + + for identifier in paper.get("identifiers", []): + if ( + pid_of(identifier.get("creator")) == "ror:01fyxcz70" + or identifier.get("schema_type") == "dlthings:DOI" + ): + return identifier.get("notation") + + +def pid_of(x: str | dict) -> str: + """Return a PID of an object, inlined or not + + A shortcut - makes a pid string or an inlined dict (where pid is a + property) equivalent. Does not do further validation, but it could + be added here. + + """ + return x.get("pid", "") if isinstance(x, dict) else x + + +def process_orcid(person: dict) -> str | None: + """Return an ORCID from identifiers. If no ORCID is found, try a ZB issued + ID.""" + + for identifier in person.get("identifiers", []): + # this is a deviation from enrich-via-doi, which filtered based on + # '("creator")) == "ror:04fa4r544"' -- in the psyinf pool, only a + # fraction of identifiers have this creator annotation + if pid_of(identifier.get("schema_type")) == "xyzri:ORCID": + return identifier.get("notation") + # if we do not find an ORCID, we make another attempt based on ZB-issued + # ID + for identifier in person.get("identifiers", []): + if pid_of(identifier.get("schema_type")) == "dlthings:Identifier" and \ + pid_of(identifier.get("creator")) == "https://w3id.org/isil/DE-Juel1": + return identifier.get("notation") + + + +map_pof_to_pid = { + 'G:(DE-HGF)POF4-5251': 'xyzrins:projects/45e542dd-2d2a-41d0-8399-90f2fde72a21', + 'G:(DE-HGF)POF4-5252': 'xyzrins:projects/858d0ddb-3c51-4cea-a0e5-a73224b59ae5', + 'G:(DE-HGF)POF4-5253': 'xyzrins:projects/64777d99-26f6-441c-9849-ba8018571de8', + 'G:(DE-HGF)POF4-5254': 'xyzrins:projects/5340dd43-6ef0-4cf8-a936-67c2a9f3a725', + 'G:(DE-HGF)POF4-5255': 'xyzrins:projects/a3f0a9e0-c945-4e04-a698-be426a9ac075', +} + +# TODO: joint work in the pool to get and create necessary infos +map_grant_to_pid = { + 'G:(EU-Grant)101147319': { + 'pid': 'xyzrins:grants/ebrains-2.0', + 'start': '2024-01-01', + 'end': '2026-12-01'}, + 'G:(GEPRIS)552122525': { #SFB B06 + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/143b2417-20a6-45d1-8140-dfddceb61749', + 'start': '2025-01-01', + 'end': '2028-12-31' + }, + 'G:(GEPRIS)458705014': { # SFB Z03 + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/1325c9f4-279a-42d1-ba37-ed5577965620', + 'start': '2021-01-01', + 'end': '2028-12-31' + }, + 'G:(GEPRIS)458684554': { # SFB C05 + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/3977dc9e-59e9-4ff2-bcbf-5bcb8f1c5c2d', + 'start': '2021-01-01', + 'end': '2028-12-31' + }, + 'G:(GEPRIS)458640473': { # SFB B05 + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/66cbbef2-5fd4-42c6-8348-cb74af435b75', + 'start': '2021-01-01', + 'end': '2028-12-31' + }, + 'G:(DE-HGF)InterLabs-0015': { # HIBALL + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/391de57c-8040-4cf7-a515-a6791b9a0c89', + 'start': '2020-04-01', + 'end': '2025-03-31' + }, + 'G:(EU-Grant)945539': { # HBP SGA3 + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/hbp-sga3', + 'start': '2020-04-01', + 'end': '2023-09-30' + }, + 'G:(EU-Grant)604102': { # HBP + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/ff4d1947-f1ad-4e9f-88fc-57af7b7cfc85', + 'start': '2013-10-01', + 'end': '2017-02-28' + }, + 'G:(GEPRIS)431549029': { # SFB 1451 insgesamt + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/sfb1451', + 'start': '2021-01-01', + 'end': '2028-12-31' + }, + 'G:(DE-Juel1)JL SMHB-2021-2027': { # JL SMHB + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/2b40e9ef-d9d2-42f8-a502-2e6efa86a1f7', + 'start': '2021-01-01', + 'end': '2027-12-31' + }, + 'G:(GEPRIS)524408221': { # Mikrostrukturelle Entwicklung des Gehirns + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/003e63be-a64b-4d72-a912-392b40710a48', + 'start': '2024-10-01', + 'end': '2027-09-30' + }, + 'G:(EU-Grant)826421': { # VirtualBrainCloud + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/4919df79-bb48-443a-b74d-3eb83e466187', + 'start': '2018-12-01', + 'end': '2023-05-31' + }, + 'G:(MKW NRW)KP22-106A': { # ABCD-J + 'pid': 'https://concepts.datalad.org/s/demo-research-information/ns/grants/mkw-abcd-j', + 'start': '2023-05-01', + 'end': '2027-04-30', + } +} + +map_content_to_bibitem = { + 'article': 'bibo:AcademicArticle', + 'lecture': 'fabio:Presentation', + 'Conference Paper': 'bibo:Proceedings', + 'CONFERENCE_PAPER': 'bibo:Proceedings', + 'PATENT': 'bibo:Patent', + 'BOOK_CHAPTER': 'bibo:Chapter', + 'Preprint': 'bibo:Manuscript', + 'Output Types/Dissertation': 'bibo:Thesis', +} + + +@click.command() +@click.option('--dtc-api-url', '-a', + default='https://pool.psychoinformatics.de/api') +@click.option('--dtc-collection', '-c', default='public') +@click.option('--file', '-f') +@click.option('--persons', type=click.File("rb"), default='.cache/Person.jsonl') +@click.option('--publications', type=click.File("rb"), + default='.cache/Publications.jsonl') +@click.option('--submit', + type=click.Choice(['XYZPerson', 'XYZPublication', 'both']), + default='both') +def main( + file: str, + dtc_api_url: str, + dtc_collection: str, + submit: str, + persons, + publications, +) -> None: + """ + Given an XML file with MARCXML publication data provided as --file, + parse publication records and submit them to a pool at --dtc-api-url + as Publication records, into the collection determined by + --dtc-collection. Set the environment variables JUSER_PW and JUSER_USER + to your Juser Login credentials. + """ + if environ.get('DTC_TOKEN', None) is None: + print("DTC_TOKEN required in environment! Aborting.") + return + if environ.get('JUSER_PW', None) is None or \ + environ.get('JUSER_USER', None) is None: + print("JUSER_PW and JUSER_USER required in environment! Aborting.") + return + all_people = [json.loads(line) for line in + persons] if persons is not None else [] + all_pubs = [json.loads(line) for line in + publications] if persons is not None else [] + JS = JuserScraper( + pool=dtc_api_url, + collection=dtc_collection, + xml=file, + persons=all_people, + pubs=all_pubs, + ) + JS.create_records() + if submit in ['both', 'XYZPublication']: + JS.submit_records( + JS.to_submit_public, + collection='public', + class_name='XYZPublication' + ) + if submit in ['both', 'XYZPerson']: + JS.submit_records( + JS.to_submit_protected, + collection='protected', + class_name='XYZPerson' + ) + +if __name__ == '__main__': + main()