From f30f22ecbd68f89076185f6743a13a60d65cb44f Mon Sep 17 00:00:00 2001 From: Debora Candela Date: Sun, 1 Mar 2026 23:18:15 +0100 Subject: [PATCH] First try Dalia-Debora --- script.sh | 58 +++++++++++++++++++++++++++---------------------------- 1 file changed, 29 insertions(+), 29 deletions(-) diff --git a/script.sh b/script.sh index 4a708d8..5c2825c 100755 --- a/script.sh +++ b/script.sh @@ -20,26 +20,29 @@ echo "" # 0. Tell me who worked on this together -echo "student 1" # please fill in names here -echo "student 2" +echo "Dalia Lemmi" # please fill in names here +echo "Debora Candela" # here is a list of tasks for you. # whenever a line says "don't touch" then you are not supposed to touch what comes below. # all lines where you need to take action are numbered. # 1. Go to your home directory: +cd -# 2. from your home, creating a directory structure: new folder `programming-hw`, and inside that folder create folder `hw1` +# 2. from your home, creating a directory structure: new folder programming-hw, and inside that folder create folder hw1 +mkdir -p ~/programming-hw/hw1 # 3. go into that new directory, i.e. into ~/programming-hw/hw1 +cd ~/programming-hw/hw1 # checking the folder exists now # don't touch [ -d ~/programming-hw/hw1 ] && echo "directory created successfully" || exit 1 # download with wget if file does not exist yet -# if wget does not work for you, manually download from the below URL and place into `~/programming-hw/hw1` as `movies.dat` +# if wget does not work for you, manually download from the below URL and place into ~/programming-hw/hw1 as movies.dat # first let me check whether you have wget installed and install it otherwise for you: # don't touch command -v wget || { [[ "$OSTYPE" == "darwin"* ]] && brew install wget || sudo apt-get install -y wget; } @@ -59,11 +62,15 @@ if [ ! -f movies.dat ]; then exit 1 fi -# 4. look at first 4 rows of downloaded data in `movies.dat`. look at this output and try to understand how it is structured. the file ending is `dat`. however, how could you also denominate such a file? +# 4. look at first 4 rows of downloaded data in movies.dat. look at this output and try to understand how it is structured. the file ending is dat. however, how could you also denominate such a file? +head -n 4 movies.dat +#looking at the output, we can see that the file is structured in 3 columns, separated by :: +#The first column is a movie ID, the second column is the movie title and the third column is the genres of the movie, separated by | if there are more than one genres +#Such a file is a csv file, because it is a "character separated values" file, where the character is :: instead of ,''' -# 5. look at first 4 rows of downloaded data in `movies.dat` redirect to a file called `first4.txt` - +# 5. look at first 4 rows of downloaded data in movies.dat redirect to a file called first4.txt +head -n 4 movies.dat > first4.txt # don't touch if [ ! -f first4.txt ]; then @@ -78,7 +85,7 @@ echo "" # actual analysis task: A pipeline # we want to know how many genres each movie is classified into -# `genre1|genre2` means it's in genre1 and genre2: we would count `2` for such an entry +# genre1|genre2 means it's in genre1 and genre2: we would count 2 for such an entry # the end product of our pipeline is a contingency table, informing us # about how many movies are part of how many genres. it would look similar to # 2 0 @@ -88,27 +95,28 @@ echo "" # I want you to construct a pipeline. let's build it up from the start -# 6. use the `awk` command to separt each row at the `::` delimters -# fill in for _filename_ the correct file you want to operate on. +# 6. use the awk command to separt each row at the :: delimters +# fill in for filename the correct file you want to operate on. # then remove the # character from the start of the line and look at the result -# awk -F '::' '{print $3}' _filename_ +awk -F '::' '{print $3}' movies.dat -# 7. observe that the `{print $3}` part prints the third field. that looks like: genre1|genre2 -# that is, there is *another* separator in this column, `|`. Let's separate again. copy your command from above and -# add a pipe as follows. here, the second statement will split at `|` and print into *how many parts* it has split. -# i.e. it will tell us *how many genres* that movie belonged to. No need to understand the `awk` part. -# again, remove the # below, fill in for _filename_ and run +# 7. observe that the {print $3} part prints the third field. that looks like: genre1|genre2 +# that is, there is another separator in this column, |. Let's separate again. copy your command from above and +# add a pipe as follows. here, the second statement will split at | and print into how many parts it has split. +# i.e. it will tell us how many genres that movie belonged to. No need to understand the awk part. +# again, remove the # below, fill in for filename and run -# awk -F '::' '{print $3}' _filename_ | awk '{print split($0, a, "\\|")}' +awk -F '::' '{print $3}' movies.dat | awk '{print split($0, a, "\\|")}' # 8. finish the pipeline by adding 2 commands, exactly like in class, that will produce a contingency table # we want to know how many movies belong to 0,1,2,... etc genres. -# awk -F '::' '{print $3}' _filename_ | awk '{print split($0, a, "\\|")}' | sort | uniq -c +awk -F '::' '{print $3}' movies.dat | awk '{print split($0, a, "\\|")}' | sort | uniq -c -# 9. redirect (>) the output of your pipeline to a file `outtable.txt` in the current directory +# 9. redirect (>) the output of your pipeline to a file outtable.txt in the current directory +awk -F '::' '{print $3}' movies.dat | awk '{print split($0, a, "\\|")}' | sort | uniq -c > outtable.txt # dont touch @@ -116,7 +124,7 @@ echo "" # leave this untouched echo "here is my table:" # this as well # 10. Print your table to screen - +cat outtable.txt #### End of your tasks @@ -133,12 +141,4 @@ else echo "" echo "wrong result :-(" exit 1 -fi - - - - - - - - +fi \ No newline at end of file