some fix to make workflows runs

2021-03-17 12:12:56 +01:00 · 2021-03-17 12:12:56 +01:00 · cc5bbafa5d
parent 9cac6da9bd
commit cc5bbafa5d
7 changed files with 92 additions and 57 deletions
--- a/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/crossref/Crossref2Oaf.scala
+++ b/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/crossref/Crossref2Oaf.scala
@ -180,7 +180,7 @@ case object Crossref2Oaf {
    // Ticket #6281 added pid to Instance
-    instance.setPid(result.getPid.asScala.filter(p => p.getQualifier.getClassid.equalsIgnoreCase("doi")).asJava)
+    instance.setPid(result.getPid)
    val has_review = (json \ "relation" \"has-review" \ "id")
--- a/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/mag/MagDataModel.scala
+++ b/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/mag/MagDataModel.scala
@ -172,7 +172,7 @@ case object ConversionUtil {
      i.setUrl(List(s"https://academic.microsoft.com/#/detail/${extractMagIdentifier(pub.getOriginalId.asScala)}").asJava)
    // Ticket #6281 added pid to Instance
-    i.setPid(pub.getPid.asScala.filter(p => p.getQualifier.getClassid.equalsIgnoreCase("doi")).asJava)
+    i.setPid(pub.getPid)
    i.setCollectedfrom(createMAGCollectedFrom())
    pub.setInstance(List(i).asJava)
--- a/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/mag/SparkProcessMAG.scala
+++ b/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/mag/SparkProcessMAG.scala
@ -67,7 +67,7 @@ object SparkProcessMAG {
          MagPaperAuthorDenormalized(mpa.PaperId, mpa.author, af.DisplayName, mpa.sequenceNumber)
        } else
          mpa
-      }).groupBy("PaperId").agg(collect_list(struct($"author", $"affiliation")).as("authors"))
+      }).groupBy("PaperId").agg(collect_list(struct($"author", $"affiliation", $"sequenceNumber")).as("authors"))
      .write.mode(SaveMode.Overwrite).save(s"$workingPath/merge_step_1_paper_authors")
    logger.info("Phase 4) create First Version of publication Entity with Paper Journal and Authors")
--- a/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/uw/UnpayWallToOAF.scala
+++ b/dhp-workflows/dhp-doiboost/src/main/java/eu/dnetlib/doiboost/uw/UnpayWallToOAF.scala
@ -65,11 +65,7 @@ object UnpayWallToOAF {
    val colour = get_color(is_oa, oaLocation, journal_is_oa)
    pub.setPid(List(createSP(doi, "doi", PID_TYPES)).asJava)
-    //IMPORTANT
+
    //The old method pub.setId(IdentifierFactory.createIdentifier(pub))
    //will be replaced using IdentifierFactory
    //pub.setId(generateIdentifier(pub, doi.toLowerCase))
    pub.setId(IdentifierFactory.createIdentifier(pub))
    pub.setCollectedfrom(List(createUnpayWallCollectedFrom()).asJava)
@ -87,7 +83,7 @@ object UnpayWallToOAF {
    i.setUrl(List(oaLocation.url.get).asJava)
    // Ticket #6281 added pid to Instance
-    i.setPid(pub.getPid.asScala.filter(p => p.getQualifier.getClassid.equalsIgnoreCase("doi")).asJava)
+    i.setPid(pub.getPid)
    if (oaLocation.license.isDefined)
      i.setLicense(asField(oaLocation.license.get))
@ -104,6 +100,14 @@ object UnpayWallToOAF {
      i.setAccessright(a)
    }
    pub.setInstance(List(i).asJava)
    //IMPORTANT
    //The old method pub.setId(IdentifierFactory.createIdentifier(pub))
    //will be replaced using IdentifierFactory
    //pub.setId(generateIdentifier(pub, doi.toLowerCase))
    val  id = IdentifierFactory.createIdentifier(pub)
    logger.info(id);
    pub.setId(IdentifierFactory.createIdentifier(pub))
    pub
  }
--- a/dhp-workflows/dhp-doiboost/src/main/resources/eu/dnetlib/dhp/doiboost/oozie_app/workflow.xml
+++ b/dhp-workflows/dhp-doiboost/src/main/resources/eu/dnetlib/dhp/doiboost/oozie_app/workflow.xml
@ -54,6 +54,11 @@
        </property>
        <!--    MAG Parameters    -->
        <property>
            <name>MAGDumpPath</name>
            <description>the MAG dump working path</description>
        </property>
        <property>
            <name>inputPathMAG</name>
            <description>the MAG working path</description>
@ -132,7 +137,10 @@
                    --executor-cores=${sparkExecutorCores}
                    --driver-memory=${sparkDriverMemory}
                    --conf spark.sql.shuffle.partitions=3840
-                    ${sparkExtraOPT}
+                    --conf spark.extraListeners=${spark2ExtraListeners}
                    --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                    --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                    --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
                </spark-opts>
                <arg>--workingPath</arg><arg>${inputPathCrossref}</arg>
                <arg>--master</arg><arg>yarn-cluster</arg>
@ -147,6 +155,43 @@
            <move source="${inputPathCrossref}/crossref_ds_updated"
                  target="${inputPathCrossref}/crossref_ds"/>
        </fs>
        <ok to="ResetMagWorkingPath"/>
        <error to="Kill"/>
    </action>
    <!-- MAG SECTION -->
    <action name="ResetMagWorkingPath">
        <fs>
            <delete path="${inputPathMAG}/dataset"/>
            <delete path="${inputPathMAG}/process"/>
        </fs>
        <ok to="ConvertMagToDataset"/>
        <error to="Kill"/>
    </action>
    <action name="ConvertMagToDataset">
        <spark xmlns="uri:oozie:spark-action:0.2">
            <master>yarn-cluster</master>
            <mode>cluster</mode>
            <name>Convert Mag to Dataset</name>
            <class>eu.dnetlib.doiboost.mag.SparkImportMagIntoDataset</class>
            <jar>dhp-doiboost-${projectVersion}.jar</jar>
            <spark-opts>
                --executor-memory=${sparkExecutorMemory}
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${MAGDumpPath}</arg>
            <arg>--targetPath</arg><arg>${inputPathMAG}/dataset</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
        <ok to="ConvertCrossrefToOAF"/>
        <error to="Kill"/>
    </action>
@ -164,46 +209,15 @@
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${inputPathCrossref}/crossref_ds</arg>
            <arg>--targetPath</arg><arg>${workingPath}</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
        <ok to="ResetMagWorkingPath"/>
        <error to="Kill"/>
    </action>
    <!-- MAG SECTION -->
    <action name="ResetMagWorkingPath">
        <fs>
            <delete path="${inputPathMAG}/dataset"/>
            <delete path="${inputPathMAG}/process"/>
            <delete path="${inputPathMAG}/dataset"/>
        </fs>
        <ok to="ConvertMagToDataset"/>
        <error to="Kill"/>
    </action>
    <action name="ConvertMagToDataset">
        <spark xmlns="uri:oozie:spark-action:0.2">
            <master>yarn-cluster</master>
            <mode>cluster</mode>
            <name>Convert Mag to Dataset</name>
            <class>eu.dnetlib.doiboost.mag.SparkImportMagIntoDataset</class>
            <jar>dhp-doiboost-${projectVersion}.jar</jar>
            <spark-opts>
                --executor-memory=${sparkExecutorMemory}
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                ${sparkExtraOPT}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${inputPathMAG}/input</arg>
            <arg>--targetPath</arg><arg>${inputPathMAG}/dataset</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
        <ok to="ProcessMAG"/>
        <error to="Kill"/>
    </action>
@ -216,11 +230,14 @@
            <class>eu.dnetlib.doiboost.mag.SparkProcessMAG</class>
            <jar>dhp-doiboost-${projectVersion}.jar</jar>
            <spark-opts>
-                --executor-memory=${sparkExecutorMemory}
+                --executor-memory=${sparkExecutorIntersectionMemory}
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${inputPathMAG}/dataset</arg>
            <arg>--workingPath</arg><arg>${inputPathMAG}/process</arg>
@ -245,10 +262,14 @@
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.sql.shuffle.partitions=3840
                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${inputPathUnpayWall}/uw_extracted</arg>
-            <arg>--targetPath</arg><arg>${workingPath}</arg>
+            <arg>--targetPath</arg><arg>${workingPath}/uwPublication</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
        <ok to="ProcessORCID"/>
@ -268,10 +289,13 @@
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--sourcePath</arg><arg>${inputPathOrcid}</arg>
-            <arg>--targetPath</arg><arg>${workingPath}</arg>
+            <arg>--targetPath</arg><arg>${workingPath}/orcidPublication</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
        <ok to="CreateDOIBoost"/>
@ -291,11 +315,15 @@
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.sql.shuffle.partitions=3840
                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--hostedByMapPath</arg><arg>${hostedByMapPath}</arg>
-            <arg>--affiliationPath</arg><arg>${inputPathMAG}/process/Affiliations</arg>
+            <arg>--affiliationPath</arg><arg>${inputPathMAG}/dataset/Affiliations</arg>
-            <arg>--paperAffiliationPath</arg><arg>${inputPathMAG}/process/PaperAuthorAffiliations</arg>
+            <arg>--paperAffiliationPath</arg><arg>${inputPathMAG}/dataset/PaperAuthorAffiliations</arg>
            <arg>--workingPath</arg><arg>${workingPath}</arg>
            <arg>--master</arg><arg>yarn-cluster</arg>
        </spark>
@ -316,7 +344,10 @@
                --executor-cores=${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.sql.shuffle.partitions=3840
-                ${sparkExtraOPT}
+                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
            </spark-opts>
            <arg>--dbPublicationPath</arg><arg>${workingPath}/doiBoostPublicationFiltered</arg>
            <arg>--dbDatasetPath</arg><arg>${workingPath}/crossrefDataset</arg>
--- a/dhp-workflows/dhp-doiboost/src/test/java/eu/dnetlib/doiboost/uw/UnpayWallMappingTest.scala
+++ b/dhp-workflows/dhp-doiboost/src/test/java/eu/dnetlib/doiboost/uw/UnpayWallMappingTest.scala
@ -28,7 +28,7 @@ class UnpayWallMappingTest {
      if(p!= null) {
        assertTrue(p.getPid.size()==1)
-        logger.info(p.getId)
+        logger.info("ID :",p.getId)
      }
      assertNotNull(line)
      assertTrue(line.nonEmpty)