fixed implementation of PatchRelationsApplication, refined the relative unit test

2021-07-30 11:07:09 +02:00 · 2021-07-30 11:07:09 +02:00 · 4f78565c04
parent a6a38cca9e 9bc4fd3b69
commit 4f78565c04
2 changed files with 40 additions and 39 deletions
--- a/dhp-workflows/dhp-graph-mapper/src/main/resources/eu/dnetlib/dhp/oa/graph/raw_all/oozie_app/workflow.xml
+++ b/dhp-workflows/dhp-graph-mapper/src/main/resources/eu/dnetlib/dhp/oa/graph/raw_all/oozie_app/workflow.xml
@ -548,42 +548,7 @@
        <error to="Kill"/>
    </action>
-    <join name="wait_graphs" to="patchRelations"/>
+    <join name="wait_graphs" to="fork_merge_claims"/>
    <decision name="decisionPatchRelations">
        <switch>
            <case to="patchRelations">
                ${(shouldPatchRelations eq "true") and
                (fs:exists(concat(concat(wf:conf('nameNode'),'/'),wf:conf('idMappingPath'))) eq "true")}
            </case>
            <default to="fork_merge_claims"/>
        </switch>
    </decision>
    <action name="patchRelations">
        <spark xmlns="uri:oozie:spark-action:0.2">
            <master>yarn</master>
            <mode>cluster</mode>
            <name>PatchRelations</name>
            <class>eu.dnetlib.dhp.oa.graph.raw.PatchRelationsApplication</class>
            <jar>dhp-graph-mapper-${projectVersion}.jar</jar>
            <spark-opts>
                --executor-memory ${sparkExecutorMemory}
                --executor-cores ${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
                --conf spark.sql.shuffle.partitions=7680
            </spark-opts>
            <arg>--graphBasePath</arg><arg>${workingDir}/graph_raw</arg>
            <arg>--workingDir</arg><arg>${workingDir}/patch_relations</arg>
            <arg>--idMappingPath</arg><arg>${idMappingPath}</arg>
        </spark>
        <ok to="fork_merge_claims"/>
        <error to="Kill"/>
    </action>
    <fork name="fork_merge_claims">
        <path start="merge_claims_publication"/>
@ -596,7 +561,6 @@
        <path start="merge_claims_relation"/>
    </fork>
    <action name="merge_claims_publication">
        <spark xmlns="uri:oozie:spark-action:0.2">
            <master>yarn</master>
@ -805,7 +769,42 @@
        <error to="Kill"/>
    </action>
-    <join name="wait_merge" to="End"/>
+    <join name="wait_merge" to="decisionPatchRelations"/>
    <decision name="decisionPatchRelations">
        <switch>
            <case to="patchRelations">
                ${(shouldPatchRelations eq "true") and
                (fs:exists(concat(concat(wf:conf('nameNode'),'/'),wf:conf('idMappingPath'))) eq "true")}
            </case>
            <default to="End"/>
        </switch>
    </decision>
    <action name="patchRelations">
        <spark xmlns="uri:oozie:spark-action:0.2">
            <master>yarn</master>
            <mode>cluster</mode>
            <name>PatchRelations</name>
            <class>eu.dnetlib.dhp.oa.graph.raw.PatchRelationsApplication</class>
            <jar>dhp-graph-mapper-${projectVersion}.jar</jar>
            <spark-opts>
                --executor-memory ${sparkExecutorMemory}
                --executor-cores ${sparkExecutorCores}
                --driver-memory=${sparkDriverMemory}
                --conf spark.extraListeners=${spark2ExtraListeners}
                --conf spark.sql.queryExecutionListeners=${spark2SqlQueryExecutionListeners}
                --conf spark.yarn.historyServer.address=${spark2YarnHistoryServerAddress}
                --conf spark.eventLog.dir=${nameNode}${spark2EventLogDir}
                --conf spark.sql.shuffle.partitions=7680
            </spark-opts>
            <arg>--graphBasePath</arg><arg>${graphOutputPath}</arg>
            <arg>--workingDir</arg><arg>${workingDir}/patch_relations</arg>
            <arg>--idMappingPath</arg><arg>${idMappingPath}</arg>
        </spark>
        <ok to="End"/>
        <error to="Kill"/>
    </action>
    <end name="End"/>
 </workflow-app>
--- a/dhp-workflows/dhp-graph-provision/src/main/java/eu/dnetlib/dhp/oa/provision/PrepareRelationsJob.java
+++ b/dhp-workflows/dhp-graph-provision/src/main/java/eu/dnetlib/dhp/oa/provision/PrepareRelationsJob.java
@ -10,6 +10,7 @@ import java.util.Set;
 import java.util.stream.Collectors;
 import org.apache.commons.io.IOUtils;
 import org.apache.commons.lang3.StringUtils;
 import org.apache.spark.SparkConf;
 import org.apache.spark.api.java.JavaRDD;
 import org.apache.spark.api.java.JavaSparkContext;
@ -81,6 +82,7 @@ public class PrepareRelationsJob {
 		Set<String> relationFilter = Optional
 			.ofNullable(parser.get("relationFilter"))
 			.map(String::toLowerCase)
 			.map(s -> Sets.newHashSet(Splitter.on(",").split(s)))
 			.orElse(new HashSet<>());
 		log.info("relationFilter: {}", relationFilter);
@ -130,7 +132,7 @@ public class PrepareRelationsJob {
 		JavaRDD<Relation> rels = readPathRelationRDD(spark, inputRelationsPath)
 			.filter(rel -> rel.getDataInfo().getDeletedbyinference() == false)
-			.filter(rel -> relationFilter.contains(rel.getRelClass()) == false);
+			.filter(rel -> relationFilter.contains(StringUtils.lowerCase(rel.getRelClass())) == false);
 		JavaRDD<Relation> pruned = pruneRels(
 			pruneRels(