CBD Prac spark
Uploaded movie_mini to HDFS: /home/sh/Uni/cbd/pr/7/movie_mini.txt
Knives Out,Comedy|Crime|Drama
Parasite,Comedy|Drama|Thriller
Uncut Gems,Crime|Drama|Thriller
Interstellar,Adventure|Drama|Sci-Fi
RDD Programming Guide - Spark 3.4.1 Documentation
sudo docker exec -it spark-master bash
>>> movies = sc.textFile("hdfs://namenode:9000/user/root/movies.dat")
>>> movies.count()
10681
>>> mov = movies.map(lambda x: x.split("::"))
>>> mov = mov.map(lambda x: x[1:])
>>> mov = mov.map(lambda x: [x[0],x[1].split("|")])
>>> mov.take(1)
'Toy Story (1995)', ['Adventure', 'Animation', 'Children', 'Comedy', 'Fantasy']
# Show movies with genre Children
>>> mov.filter(lambda x: True if 'Children' in x[1] else False).take(5)
'Toy Story (1995)', ['Adventure', 'Animation', 'Children', 'Comedy', 'Fantasy', ['Jumanji (1995)', ['Adventure', 'Children', 'Fantasy']], ['Tom and Huck (1995)', ['Adventure', 'Children']], ['Balto (1995)', ['Animation', 'Children']], ['Babe (1995)', ['Children', 'Comedy', 'Drama', 'Fantasy']]]
>>> mov.filter(lambda x: True if 'Children' in x[1] else False).count()
528
>>> children = mov.filter(lambda x: True if 'Children' in x[1] else False)
>>> children.saveAsTextFile("hdfs://namenode:9000/user/root/movies_children2.dat")
>>> genre_counts = mov.flatMap(lambda x: x[1]).map(lambda x: (x,1)).reduceByKey(lambda a,b: a+b)
>>> genre_counts.take(4)
[('Horror', 1013), ('Sci-Fi', 754), ('Drama', 5339), ('Documentary', 482)]
>>> genre_counts.sortBy(lambda x: x[1], False).take(10)
[('Drama', 5339), ('Comedy', 3703), ('Thriller', 1706), ('Romance', 1685), ('Action', 1473), ('Crime', 1118), ('Adventure', 1025), ('Horror', 1013), ('Sci-Fi', 754), ('Fantasy', 543)]
>>> genre_counts.sortBy(lambda x: x[1], False).toDF().show()
+------------------+----+
| _1| _2|
+------------------+----+
| Drama|5339|
| Comedy|3703|
| Thriller|1706|
| Romance|1685|
| Action|1473|
| Crime|1118|
| Adventure|1025|
| Horror|1013|
| Sci-Fi| 754|
| Fantasy| 543|
| Children| 528|
| War| 511|
| Mystery| 509|
| Documentary| 482|
| Musical| 436|
| Animation| 286|
| Western| 275|
| Film-Noir| 148|
| IMAX| 29|
|(no genres listed)| 1|
+------------------+----+
Cool things:
# Pretty output
mov.toDF().show()
help(mov) -> interactive help on all things RDD can do!
# Parsing the year
>>> mov.map(lambda x: x[0].split("(")[-1][:-1]).take(4)
['1995', '1995', '1995', '1995']
# (name, year)
>>> mov.map(lambda x: (''.join(x[0].split("(")[:-1]),x[0].split("(")[-1][:-1])).take(40)
[('Toy Story ', '1995'), ('Jumanji ', '1995'), ('Grumpier Old Men ', '1995'), ('Waiting to Exhale ', '1995'), ('Father of the Bride Part II ', '1995'), ('Heat ', '1995'), ('Sabrina ', '1995'), ('Tom and Huck ', '1995'), ('Sudden Death ', '1995'), ('GoldenEye ', '1995'), ('American President, The ', '1995'), ('Dracula: Dead and Loving It ', '1995'), ('Balto ', '1995'), ('Nixon ', '1995'), ('Cutthroat Island ', '1995'), ('Casino ', '1995'), ('Sense and Sensibility ', '1995'), ('Four Rooms ', '1995'), ('Ace Ventura: When Nature Calls ', '1995'), ('Money Train ', '1995'), ('Get Shorty ', '1995'), ('Copycat ', '1995'), ('Assassins ', '1995'), ('Powder ', '1995'), ('Leaving Las Vegas ', '1995'), ('Othello ', '1995'), ('Now and Then ', '1995'), ('Persuasion ', '1995'), ('City of Lost Children, The Cité des enfants perdus, La) ', '1995'), ('Shanghai Triad Yao a yao yao dao waipo qiao) ', '1995'), ('Dangerous Minds ', '1995'), ('12 MonkeysTwelve Monkeys) ', '1995'), ('Wings of Courage ', '1995'), ('Babe ', '1995'), ('Carrington ', '1995'), ('Dead Man Walking ', '1995'), ('Across the Sea of Time ', '1995'), ('It Takes Two ', '1995'), ('Clueless ', '1995'), ('Cry, the Beloved Country ', '1995')]
# the rest
>>> mov.map(lambda x: (''.join(x[0].split("(")[:-1]),x[0].split("(")[-1][:-1],"|".join(x[1]))).take(4)
[('Toy Story ', '1995', 'Adventure|Animation|Children|Comedy|Fantasy'), ('Jumanji ', '1995', 'Adventure|Children|Fantasy'), ('Grumpier Old Men ', '1995', 'Comedy|Romance'), ('Waiting to Exhale ', '1995', 'Comedy|Drama|Romance')]
>>> m = mov.map(lambda x: (''.join(x[0].split("(")[:-1]),x[0].split("(")[-1][:-1],"|".join(x[1])))>>> m.toDF().show()
+--------------------+----+--------------------+
| _1| _2| _3|
+--------------------+----+--------------------+
| Toy Story |1995|Adventure|Animati...|
| Jumanji |1995|Adventure|Childre...|
| Grumpier Old Men |1995| Comedy|Romance|
| Waiting to Exhale |1995|Comedy|Drama|Romance|
|Father of the Bri...|1995| Comedy|
| Heat |1995|Action|Crime|Thri...|
| Sabrina |1995| Comedy|Romance|
| Tom and Huck |1995| Adventure|Children|
| Sudden Death |1995| Action|
| GoldenEye |1995|Action|Adventure|...|
|American Presiden...|1995|Comedy|Drama|Romance|
|Dracula: Dead and...|1995| Comedy|Horror|
| Balto |1995| Animation|Children|
| Nixon |1995| Drama|
| Cutthroat Island |1995|Action|Adventure|...|
| Casino |1995| Crime|Drama|
|Sense and Sensibi...|1995|Comedy|Drama|Romance|
| Four Rooms |1995|Comedy|Drama|Thri...|
|Ace Ventura: When...|1995| Comedy|
| Money Train |1995|Action|Comedy|Cri...|
+--------------------+----+--------------------+
only showing top 20 rows
>>> m.map(lambda x: (x[1],1)).reduceByKey(lambda a,b: a+b).sortBy(lambda x: x[1], False).toDF().show()
+----+---+
| _1| _2|
+----+---+
|2002|441|
|2000|405|
|2001|403|
|1998|384|
|1996|384|
|1997|370|
|2003|366|
|2007|364|
|1995|362|
|1999|357|
|2006|345|
|2004|342|
|2005|332|
|1994|307|
|1993|258|
|2008|251|
|1988|214|
|1989|212|
|1992|212|
|1987|205|
+----+---+
only showing top 20 rows
>>> m.map(lambda x: (x[1],1)).reduceByKey(lambda a,b: a+b).sortBy(lambda x: x[1], False).collect()
[('2002', 441), ('2000', 405), ('2001', 403), ('1998', 384), ('1996', 384), ('1997', 370), ('2003', 366), ('2007', 364), ('1995', 362), ('1999', 357), ('2006', 345), ('2004', 342), ('2005', 332), ('1994', 307), ('1993', 258), ('2008', 251), ('1988', 214), ('1989', 212), ('1992', 212), ('1987', 205), ('1990', 200), ('1991', 188), ('1981', 178), ('1982', 170), ('1986', 166), ('1980', 161), ('1985', 158), ('1984', 137), ('1983', 111), ('1979', 87), ('1966', 87), ('1977', 83), ('1972', 83), ('1978', 82), ('1973', 81), ('1976', 75), ('1974', 75), ('1975', 74), ('1971', 73), ('1964', 72), ('1965', 72), ('1968', 72), ('1970', 71), ('1962', 69), ('1967', 68), ('1960', 66), ('1969', 64), ('1963', 63), ('1958', 62), ('1957', 62), ('1959', 61), ('1955', 57), ('1961', 57), ('1953', 55), ('1956', 53), ('1948', 46), ('1950', 44), ('1951', 44), ('1954', 43), ('1952', 40), ('1940', 40), ('1943', 40), ('1947', 39), ('1942', 38), ('1946', 38), ('1944', 37), ('1949', 37), ('1939', 37), ('1945', 36), ('1936', 32), ('1937', 30), ('1941', 28), ('1933', 23), ('1932', 22), ('1938', 19), ('1927', 19), ('1935', 18), ('1934', 18), ('1931', 16), ('1930', 15), ('1926', 10), ('1928', 10), ('1925', 10), ('1929', 7), ('1922', 7), ('1924', 6), ('1923', 6), ('1920', 5), ('1919', 4), ('1921', 3), ('1916', 2), ('1917', 2), ('1918', 2), ('1915', 1)]
dataframes!
>>> cols = ['movie_name','year','genres']
>>> df = spark.createDataFrame(m,schema=cols)
>>> df.show()
+--------------------+----+--------------------+
| movie_name|year| genres|
+--------------------+----+--------------------+
| Toy Story |1995|Adventure|Animati...|
| Jumanji |1995|Adventure|Childre...|
| Grumpier Old Men |1995| Comedy|Romance|
| Waiting to Exhale |1995|Comedy|Drama|Romance|
|Father of the Bri...|1995| Comedy|
| Heat |1995|Action|Crime|Thri...|
| Sabrina |1995| Comedy|Romance|
| Tom and Huck |1995| Adventure|Children|
| Sudden Death |1995| Action|
| GoldenEye |1995|Action|Adventure|...|
|American Presiden...|1995|Comedy|Drama|Romance|
|Dracula: Dead and...|1995| Comedy|Horror|
| Balto |1995| Animation|Children|
| Nixon |1995| Drama|
| Cutthroat Island |1995|Action|Adventure|...|
| Casino |1995| Crime|Drama|
|Sense and Sensibi...|1995|Comedy|Drama|Romance|
| Four Rooms |1995|Comedy|Drama|Thri...|
|Ace Ventura: When...|1995| Comedy|
| Money Train |1995|Action|Comedy|Cri...|
+--------------------+----+--------------------+
only showing top 20 rows
>>> df.groupBy("year").count().sort("count",ascending=False).show()
+----+-----+
|year|count|
+----+-----+
|2002| 441|
|2000| 405|
|2001| 403|
|1996| 384|
|1998| 384|
|1997| 370|
|2003| 366|
|2007| 364|
|1995| 362|
|1999| 357|
|2006| 345|
|2004| 342|
|2005| 332|
|1994| 307|
|1993| 258|
|2008| 251|
|1988| 214|
|1989| 212|
|1992| 212|
|1987| 205|
+----+-----+
only showing top 20 rows
>>> df.groupBy("year").count().sort("count",ascending=False).show(300)
+----+-----+
|year|count|
+----+-----+
|2002| 441|
|2000| 405|
|2001| 403|
|1996| 384|
|1998| 384|
|1997| 370|
|2003| 366|
|2007| 364|
|1995| 362|
|1999| 357|
|2006| 345|
|2004| 342|
|2005| 332|
|1994| 307|
|1993| 258|
|2008| 251|
|1988| 214|
|1992| 212|
|1989| 212|
|1987| 205|
|1990| 200|
|1991| 188|
|1981| 178|
|1982| 170|
|1986| 166|
|1980| 161|
|1985| 158|
|1984| 137|
|1983| 111|
|1979| 87|
|1966| 87|
|1977| 83|
|1972| 83|
|1978| 82|
|1973| 81|
|1976| 75|
|1974| 75|
|1975| 74|
|1971| 73|
|1968| 72|
|1964| 72|
|1965| 72|
|1970| 71|
|1962| 69|
|1967| 68|
|1960| 66|
|1969| 64|
|1963| 63|
|1958| 62|
|1957| 62|
|1959| 61|
|1961| 57|
|1955| 57|
|1953| 55|
|1956| 53|
|1948| 46|
|1950| 44|
|1951| 44|
|1954| 43|
|1943| 40|
|1940| 40|
|1952| 40|
|1947| 39|
|1942| 38|
|1946| 38|
|1939| 37|
|1944| 37|
|1949| 37|
|1945| 36|
|1936| 32|
|1937| 30|
|1941| 28|
|1933| 23|
|1932| 22|
|1938| 19|
|1927| 19|
|1934| 18|
|1935| 18|
|1931| 16|
|1930| 15|
|1926| 10|
|1925| 10|
|1928| 10|
|1929| 7|
|1922| 7|
|1924| 6|
|1923| 6|
|1920| 5|
|1919| 4|
|1921| 3|
|1918| 2|
|1916| 2|
|1917| 2|
|1915| 1|
+----+-----+
Nel mezzo del deserto posso dire tutto quello che voglio.
comments powered by Disqus