2020-11-06 13:47:50 +01:00
|
|
|
<workflow-app name="Gen Orcid Authors From Summaries" xmlns="uri:oozie:workflow:0.5">
|
2020-07-03 23:30:31 +02:00
|
|
|
<parameters>
|
|
|
|
<property>
|
|
|
|
<name>workingPath</name>
|
|
|
|
<description>the working dir base path</description>
|
|
|
|
</property>
|
|
|
|
<property>
|
|
|
|
<name>shell_cmd_0</name>
|
2020-11-06 13:47:50 +01:00
|
|
|
<value>wget -O /tmp/ORCID_2020_10_summaries.tar.gz https://orcid.figshare.com/ndownloader/files/25032905 ; hdfs dfs -copyFromLocal /tmp/ORCID_2020_10_summaries.tar.gz /data/orcid_activities_2020/ORCID_2020_10_summaries.tar.gz ; rm -f /tmp/ORCID_2020_10_summaries.tar.gz
|
2020-07-03 23:30:31 +02:00
|
|
|
</value>
|
|
|
|
<description>the shell command that downloads and puts to hdfs orcid summaries</description>
|
|
|
|
</property>
|
|
|
|
</parameters>
|
|
|
|
|
|
|
|
<start to="ResetWorkingPath"/>
|
|
|
|
|
|
|
|
|
|
|
|
<kill name="Kill">
|
|
|
|
<message>Action failed, error message[${wf:errorMessage(wf:lastErrorNode())}]</message>
|
|
|
|
</kill>
|
|
|
|
|
|
|
|
<action name="ResetWorkingPath">
|
|
|
|
<fs>
|
2020-11-06 13:47:50 +01:00
|
|
|
<delete path='${workingPath}/authors'/>
|
|
|
|
<mkdir path='${workingPath}/authors'/>
|
2020-07-03 23:30:31 +02:00
|
|
|
</fs>
|
|
|
|
<ok to="check_exist_on_hdfs_summaries"/>
|
|
|
|
<error to="Kill"/>
|
|
|
|
</action>
|
|
|
|
|
|
|
|
<decision name="check_exist_on_hdfs_summaries">
|
|
|
|
<switch>
|
|
|
|
<case to="ImportOrcidSummaries">
|
2020-11-06 13:47:50 +01:00
|
|
|
${fs:exists(concat(workingPath,'/ORCID_2020_10_summaries.tar.gz'))}
|
2020-07-03 23:30:31 +02:00
|
|
|
</case>
|
|
|
|
<default to="DownloadSummaries" />
|
|
|
|
</switch>
|
|
|
|
</decision>
|
|
|
|
|
|
|
|
<action name="DownloadSummaries">
|
|
|
|
<shell xmlns="uri:oozie:shell-action:0.1">
|
|
|
|
<job-tracker>${jobTracker}</job-tracker>
|
|
|
|
<name-node>${nameNode}</name-node>
|
|
|
|
<exec>bash</exec>
|
|
|
|
<argument>-c</argument>
|
|
|
|
<argument>${shell_cmd_0}</argument>
|
|
|
|
<capture-output/>
|
|
|
|
</shell>
|
|
|
|
<ok to="ImportOrcidSummaries"/>
|
|
|
|
<error to="Kill"/>
|
|
|
|
</action>
|
|
|
|
|
|
|
|
<action name="ImportOrcidSummaries">
|
|
|
|
<java>
|
|
|
|
<job-tracker>${jobTracker}</job-tracker>
|
|
|
|
<name-node>${nameNode}</name-node>
|
|
|
|
<main-class>eu.dnetlib.doiboost.orcid.OrcidDSManager</main-class>
|
|
|
|
<arg>-w</arg><arg>${workingPath}/</arg>
|
|
|
|
<arg>-n</arg><arg>${nameNode}</arg>
|
2020-11-06 13:47:50 +01:00
|
|
|
<arg>-f</arg><arg>ORCID_2020_10_summaries.tar.gz</arg>
|
|
|
|
<arg>-o</arg><arg>authors/</arg>
|
2020-07-03 23:30:31 +02:00
|
|
|
</java>
|
|
|
|
<ok to="End"/>
|
|
|
|
<error to="Kill"/>
|
|
|
|
</action>
|
|
|
|
|
|
|
|
<end name="End"/>
|
|
|
|
</workflow-app>
|