diff options
author | James Thomas <jamesjoethomas@gmail.com> | 2016-06-28 16:12:48 -0700 |
---|---|---|
committer | Tathagata Das <tathagata.das1565@gmail.com> | 2016-06-28 16:12:48 -0700 |
commit | 3554713a163c58ca176ffde87d2c6e4a91bacb50 (patch) | |
tree | 3e8e18f14eb0f1c5d00a535b17360414b1cf4fb3 /examples/src/main/python/sql/streaming | |
parent | 8a977b065418f07d2bf4fe1607a5534c32d04c47 (diff) | |
download | spark-3554713a163c58ca176ffde87d2c6e4a91bacb50.tar.gz spark-3554713a163c58ca176ffde87d2c6e4a91bacb50.tar.bz2 spark-3554713a163c58ca176ffde87d2c6e4a91bacb50.zip |
[SPARK-16114][SQL] structured streaming network word count examples
## What changes were proposed in this pull request?
Network word count example for structured streaming
## How was this patch tested?
Run locally
Author: James Thomas <jamesjoethomas@gmail.com>
Author: James Thomas <jamesthomas@Jamess-MacBook-Pro.local>
Closes #13816 from jjthomas/master.
Diffstat (limited to 'examples/src/main/python/sql/streaming')
-rw-r--r-- | examples/src/main/python/sql/streaming/structured_network_wordcount.py | 76 |
1 files changed, 76 insertions, 0 deletions
diff --git a/examples/src/main/python/sql/streaming/structured_network_wordcount.py b/examples/src/main/python/sql/streaming/structured_network_wordcount.py new file mode 100644 index 0000000000..32d63c52c9 --- /dev/null +++ b/examples/src/main/python/sql/streaming/structured_network_wordcount.py @@ -0,0 +1,76 @@ +# +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +""" + Counts words in UTF8 encoded, '\n' delimited text received from the network every second. + Usage: structured_network_wordcount.py <hostname> <port> + <hostname> and <port> describe the TCP server that Structured Streaming + would connect to receive data. + + To run this on your local machine, you need to first run a Netcat server + `$ nc -lk 9999` + and then run the example + `$ bin/spark-submit examples/src/main/python/sql/streaming/structured_network_wordcount.py + localhost 9999` +""" +from __future__ import print_function + +import sys + +from pyspark.sql import SparkSession +from pyspark.sql.functions import explode +from pyspark.sql.functions import split + +if __name__ == "__main__": + if len(sys.argv) != 3: + print("Usage: structured_network_wordcount.py <hostname> <port>", file=sys.stderr) + exit(-1) + + host = sys.argv[1] + port = int(sys.argv[2]) + + spark = SparkSession\ + .builder\ + .appName("StructuredNetworkWordCount")\ + .getOrCreate() + + # Create DataFrame representing the stream of input lines from connection to host:port + lines = spark\ + .readStream\ + .format('socket')\ + .option('host', host)\ + .option('port', port)\ + .load() + + # Split the lines into words + words = lines.select( + explode( + split(lines.value, ' ') + ).alias('word') + ) + + # Generate running word count + wordCounts = words.groupBy('word').count() + + # Start running the query that prints the running counts to the console + query = wordCounts\ + .writeStream\ + .outputMode('complete')\ + .format('console')\ + .start() + + query.awaitTermination() |