Merge pull request #18 from Java-Cesco/feature/#3

Aggregation

Merge pull request #18 from Java-Cesco/feature/#3
Aggregation
신은섭(Shin Eun Seop) · GitHub
Commit 30034ba197f88560d467d44e7fa9ff0221dde95c 30034ba1 2 parents 7a23214c dfabb771
Showing 16 changed files with 159 additions and 60 deletions
.gitignore
.idea/.name
.idea/Detecting_fraud_clicks.iml
.idea/compiler.xml
.idea/markdown-exported-files.xml
.idea/misc.xml
.idea/modules.xml
.idea/vcs.xml
2018-1-java.iml
README.md
pom.xml
src/main/java/Aggregation.java
src/main/java/MapExample.java
src/main/java/valid.java
src/test/java/testValid.java
train_sample.csv
--- a/.gitignore 100644 → 100755
View file @30034ba
+++ b/.gitignore 100644 → 100755
View file @30034ba
--- a/.idea/.name 0 → 100644
View file @30034ba
+++ b/.idea/.name 0 → 100644
View file @30034ba
+Detecting_fraud_clicks
\ No newline at end of file
--- a/.idea/Detecting_fraud_clicks.iml
View file @30034ba
+++ b/.idea/Detecting_fraud_clicks.iml
View file @30034ba
--- a/.idea/compiler.xml 0 → 100644
View file @30034ba
+++ b/.idea/compiler.xml 0 → 100644
View file @30034ba
+<?xml version="1.0" encoding="UTF-8"?>
+<project version="4">
+  <component name="CompilerConfiguration">
+    <annotationProcessing>
+      <profile name="Maven default annotation processors profile" enabled="true">
+        <sourceOutputDir name="target/generated-sources/annotations" />
+        <sourceTestOutputDir name="target/generated-test-sources/test-annotations" />
+        <outputRelativeToContentRoot value="true" />
+        <module name="Detecting_fraud_clicks" />
+      </profile>
+    </annotationProcessing>
+  </component>
+</project>
\ No newline at end of file
--- a/.idea/markdown-exported-files.xml 0 → 100644
View file @30034ba
+++ b/.idea/markdown-exported-files.xml 0 → 100644
View file @30034ba
+<?xml version="1.0" encoding="UTF-8"?>
+<project version="4">
+  <component name="MarkdownExportedFiles">
+    <htmlFiles />
+    <imageFiles />
+    <otherFiles />
+  </component>
+</project>
\ No newline at end of file
--- a/.idea/misc.xml
View file @30034ba
+++ b/.idea/misc.xml
View file @30034ba
 <?xml version="1.0" encoding="UTF-8"?>
 <project version="4">
-  <component name="JavaScriptSettings">
+  <component name="ExternalStorageConfigurationManager" enabled="true" />
-    <option name="languageLevel" value="ES6" />
+  <component name="MavenProjectsManager">
+    <option name="originalFiles">
+      <list>
+        <option value="$PROJECT_DIR$/pom.xml" />
+      </list>
+    </option>
+  </component>
+  <component name="ProjectRootManager" version="2" languageLevel="JDK_1_8" project-jdk-name="1.8" project-jdk-type="JavaSDK">
+    <output url="file://$PROJECT_DIR$/out" />
+  </component>
+  <component name="MavenProjectsManager">
+    <option name="originalFiles">
+      <list>
+        <option value="$PROJECT_DIR$/pom.xml" />
+      </list>
+    </option>
+  </component>
+  <component name="ProjectRootManager" version="2" languageLevel="JDK_1_8" default="false" project-jdk-name="1.8" project-jdk-type="JavaSDK">
+    <output url="file:///tmp" />
   </component>
 </project>
\ No newline at end of file
--- a/.idea/modules.xml deleted 100644 → 0
View file @7a23214
+++ b/.idea/modules.xml deleted 100644 → 0
View file @7a23214
-<?xml version="1.0" encoding="UTF-8"?>
-<project version="4">
-  <component name="ProjectModuleManager">
-    <modules>
-      <module fileurl="file://$PROJECT_DIR$/.idea/Detecting_fraud_clicks.iml" filepath="$PROJECT_DIR$/.idea/Detecting_fraud_clicks.iml" />
-    </modules>
-  </component>
-</project>
\ No newline at end of file
--- a/.idea/vcs.xml
View file @30034ba
+++ b/.idea/vcs.xml
View file @30034ba
 <?xml version="1.0" encoding="UTF-8"?>
 <project version="4">
   <component name="VcsDirectoryMappings">
-    <mapping directory="" vcs="Git" />
+    <mapping directory="$PROJECT_DIR$" vcs="Git" />
   </component>
 </project>
\ No newline at end of file
--- a/2018-1-java.iml 100644 → 100755
View file @30034ba
+++ b/2018-1-java.iml 100644 → 100755
View file @30034ba
--- a/README.md 100644 → 100755
View file @30034ba
+++ b/README.md 100644 → 100755
View file @30034ba
--- a/pom.xml 100644 → 100755
View file @30034ba
+++ b/pom.xml 100644 → 100755
View file @30034ba
@@ -16,7 +16,35 @@
             <artifactId>spark-core_2.11</artifactId>
             <version>2.3.0</version>
         </dependency>
-
+        <dependency>
+            <groupId>org.apache.spark</groupId>
+            <artifactId>spark-sql_2.11</artifactId>
+            <version>2.2.0</version>
+        </dependency>
+        <dependency>
+            <groupId>org.apache.spark</groupId>
+            <artifactId>spark-sql_2.11</artifactId>
+            <version>2.3.0</version>
+        </dependency>
+        <dependency>
+            <groupId>com.databricks</groupId>
+            <artifactId>spark-csv_2.11</artifactId>
+            <version>1.5.0</version>
+        </dependency>
     </dependencies>
+
+    <build>
+        <plugins>
+            <plugin>
+                <groupId>org.apache.maven.plugins</groupId>
+                <artifactId>maven-compiler-plugin</artifactId>
+                <version>3.6.1</version>
+                <configuration>
+                    <source>1.8</source>
+                    <target>1.8</target>
+                </configuration>
+            </plugin>
+        </plugins>
+    </build>
 </project>
\ No newline at end of file
--- a/src/main/java/Aggregation.java 0 → 100644
View file @30034ba
+++ b/src/main/java/Aggregation.java 0 → 100644
View file @30034ba
+import org.apache.spark.sql.Dataset;
+import org.apache.spark.sql.Row;
+import org.apache.spark.sql.SparkSession;
+import org.apache.spark.sql.expressions.Window;
+import org.apache.spark.sql.expressions.WindowSpec;
+
+import static org.apache.spark.sql.functions.*;
+import static org.apache.spark.sql.functions.lit;
+import static org.apache.spark.sql.functions.when;
+
+public class Aggregation {
+
+    public static void main(String[] args) throws Exception {
+
+        //Create Session
+        SparkSession spark = SparkSession
+                .builder()
+                .appName("Detecting Fraud Clicks")
+                .master("local")
+                .getOrCreate();
+        
+        // Aggregation
+        Aggregation agg = new Aggregation();
+        
+        Dataset<Row> dataset = agg.loadCSVDataSet("./train_sample.csv", spark);
+        dataset = agg.changeTimestempToLong(dataset);
+        dataset = agg.averageValidClickCount(dataset);
+        dataset = agg.clickTimeDelta(dataset);
+        dataset = agg.countClickInTenMinutes(dataset);
+        
+        //test
+        dataset.where("ip == '5348' and app == '19'").show(10);
+    }
+        
+        
+    private Dataset<Row> loadCSVDataSet(String path, SparkSession spark){
+        // Read SCV to DataSet
+        return spark.read().format("csv")
+                .option("inferSchema", "true")
+                .option("header", "true")
+                .load(path);
+    }
+    
+    private Dataset<Row> changeTimestempToLong(Dataset<Row> dataset){
+        // cast timestamp to long
+        Dataset<Row> newDF = dataset.withColumn("utc_click_time", dataset.col("click_time").cast("long"));
+        newDF = newDF.withColumn("utc_attributed_time", dataset.col("attributed_time").cast("long"));
+        newDF = newDF.drop("click_time").drop("attributed_time");
+        return newDF;
+    }
+         
+    private Dataset<Row> averageValidClickCount(Dataset<Row> dataset){
+        // set Window partition by 'ip' and 'app' order by 'utc_click_time' select rows between 1st row to current row
+        WindowSpec w = Window.partitionBy("ip", "app")
+                .orderBy("utc_click_time")
+                .rowsBetween(Window.unboundedPreceding(), Window.currentRow());
+
+        // aggregation
+        Dataset<Row> newDF = dataset.withColumn("cum_count_click", count("utc_click_time").over(w));
+        newDF = newDF.withColumn("cum_sum_attributed", sum("is_attributed").over(w));
+        newDF = newDF.withColumn("avg_valid_click_count", col("cum_sum_attributed").divide(col("cum_count_click")));
+        newDF = newDF.drop("cum_count_click", "cum_sum_attributed");
+        return newDF;
+    }
+
+    private Dataset<Row> clickTimeDelta(Dataset<Row> dataset){
+        WindowSpec w = Window.partitionBy ("ip")
+                .orderBy("utc_click_time");
+
+        Dataset<Row> newDF = dataset.withColumn("lag(utc_click_time)", lag("utc_click_time",1).over(w));
+        newDF = newDF.withColumn("click_time_delta", when(col("lag(utc_click_time)").isNull(),
+                lit(0)).otherwise(col("utc_click_time")).minus(when(col("lag(utc_click_time)").isNull(),
+                lit(0)).otherwise(col("lag(utc_click_time)"))));
+        newDF = newDF.drop("lag(utc_click_time)");
+        return newDF;
+    }
+    
+    private Dataset<Row> countClickInTenMinutes(Dataset<Row> dataset){
+        WindowSpec w = Window.partitionBy("ip")
+                .orderBy("utc_click_time")
+                .rangeBetween(Window.currentRow(),Window.currentRow()+600);
+
+        Dataset<Row> newDF = dataset.withColumn("count_click_in_ten_mins",
+                (count("utc_click_time").over(w)).minus(1));    //TODO 본인것 포함할 것인지 정해야함.
+        return newDF;
+    }
+}
--- a/src/main/java/MapExample.java deleted 100644 → 0
View file @7a23214
+++ b/src/main/java/MapExample.java deleted 100644 → 0
View file @7a23214
-import org.apache.spark.SparkConf;
-import org.apache.spark.api.java.JavaRDD;
-import org.apache.spark.api.java.JavaSparkContext;
-import scala.Tuple2;
-
-import java.util.Arrays;
-import java.util.List;
-
-public class MapExample {
-
-    static SparkConf conf = new SparkConf().setMaster("local[*]").setAppName("Cesco");
-    static JavaSparkContext sc = new JavaSparkContext(conf);
-    
-    public static void main(String[] args) throws Exception {
-        
-        // Parallelized with 2 partitions
-        JavaRDD<String> x = sc.parallelize(
-                Arrays.asList("spark", "rdd", "example", "sample", "example"),
-                2);
-
-        // Word Count Map Example
-        JavaRDD<Tuple2<String, Integer>> y1 = x.map(e -> new Tuple2<>(e, 1));
-        List<Tuple2<String, Integer>> list1 = y1.collect();
-
-        // Another example of making tuple with string and it's length
-        JavaRDD<Tuple2<String, Integer>> y2 = x.map(e -> new Tuple2<>(e, e.length()));
-        List<Tuple2<String, Integer>> list2 = y2.collect();
-        
-        System.out.println(list1);
-    }
-}
--- a/src/main/java/valid.java deleted 100644 → 0
View file @7a23214
+++ b/src/main/java/valid.java deleted 100644 → 0
View file @7a23214
-public class valid {
-    private int x;
-    
-    valid() {
-        x = 0;
-    }
-    
-    void printX(){
-        System.out.println(x);
-    }
-    
-    public static void main(String[] args){
-        valid v = new valid();
-        v.printX();
-    }
-    
-}
--- a/src/test/java/testValid.java 100644 → 100755
View file @30034ba
+++ b/src/test/java/testValid.java 100644 → 100755
View file @30034ba
--- a/train_sample.csv 0 → 100644
View file @30034ba
+++ b/train_sample.csv 0 → 100644
View file @30034ba