aboutsummaryrefslogtreecommitdiff
path: root/examples/src/main/python/sql/streaming
diff options
context:
space:
mode:
authorJames Thomas <jamesjoethomas@gmail.com>2016-06-28 16:12:48 -0700
committerTathagata Das <tathagata.das1565@gmail.com>2016-06-28 16:12:48 -0700
commit3554713a163c58ca176ffde87d2c6e4a91bacb50 (patch)
tree3e8e18f14eb0f1c5d00a535b17360414b1cf4fb3 /examples/src/main/python/sql/streaming
parent8a977b065418f07d2bf4fe1607a5534c32d04c47 (diff)
downloadspark-3554713a163c58ca176ffde87d2c6e4a91bacb50.tar.gz
spark-3554713a163c58ca176ffde87d2c6e4a91bacb50.tar.bz2
spark-3554713a163c58ca176ffde87d2c6e4a91bacb50.zip
[SPARK-16114][SQL] structured streaming network word count examples
## What changes were proposed in this pull request? Network word count example for structured streaming ## How was this patch tested? Run locally Author: James Thomas <jamesjoethomas@gmail.com> Author: James Thomas <jamesthomas@Jamess-MacBook-Pro.local> Closes #13816 from jjthomas/master.
Diffstat (limited to 'examples/src/main/python/sql/streaming')
-rw-r--r--examples/src/main/python/sql/streaming/structured_network_wordcount.py76
1 files changed, 76 insertions, 0 deletions
diff --git a/examples/src/main/python/sql/streaming/structured_network_wordcount.py b/examples/src/main/python/sql/streaming/structured_network_wordcount.py
new file mode 100644
index 0000000000..32d63c52c9
--- /dev/null
+++ b/examples/src/main/python/sql/streaming/structured_network_wordcount.py
@@ -0,0 +1,76 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements. See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+
+"""
+ Counts words in UTF8 encoded, '\n' delimited text received from the network every second.
+ Usage: structured_network_wordcount.py <hostname> <port>
+ <hostname> and <port> describe the TCP server that Structured Streaming
+ would connect to receive data.
+
+ To run this on your local machine, you need to first run a Netcat server
+ `$ nc -lk 9999`
+ and then run the example
+ `$ bin/spark-submit examples/src/main/python/sql/streaming/structured_network_wordcount.py
+ localhost 9999`
+"""
+from __future__ import print_function
+
+import sys
+
+from pyspark.sql import SparkSession
+from pyspark.sql.functions import explode
+from pyspark.sql.functions import split
+
+if __name__ == "__main__":
+ if len(sys.argv) != 3:
+ print("Usage: structured_network_wordcount.py <hostname> <port>", file=sys.stderr)
+ exit(-1)
+
+ host = sys.argv[1]
+ port = int(sys.argv[2])
+
+ spark = SparkSession\
+ .builder\
+ .appName("StructuredNetworkWordCount")\
+ .getOrCreate()
+
+ # Create DataFrame representing the stream of input lines from connection to host:port
+ lines = spark\
+ .readStream\
+ .format('socket')\
+ .option('host', host)\
+ .option('port', port)\
+ .load()
+
+ # Split the lines into words
+ words = lines.select(
+ explode(
+ split(lines.value, ' ')
+ ).alias('word')
+ )
+
+ # Generate running word count
+ wordCounts = words.groupBy('word').count()
+
+ # Start running the query that prints the running counts to the console
+ query = wordCounts\
+ .writeStream\
+ .outputMode('complete')\
+ .format('console')\
+ .start()
+
+ query.awaitTermination()