# Spotify_Streams_Per_Top_Artist


source ("~rsaporta/git/orch/src/Spotify_Streams_Per_Top_Artist/00 Setup.r")

                  "-------------------------------------------------"
# -------------------------------------------------------------------------------- 


{
  if (exists("DT.nms")) {
    DT.nms.bak <- copy(DT.nms)
    assign(timeStamp("DT.nms", seconds=TRUE), value=DT.nms)
  }
  DT.nms <- CJ(country=countries.avail, date=dates.avail)
  DT.nms[, nms := sprintf("%s.%s", country, date)]
  DT.nms[, url := sprintf("http://charts.spotify.com/api/tracks/most_streamed/%s/weekly/%s", country, date)]
  setkeyIfNot(DT.nms, "nms", organize=TRUE, warnForColNameInEnv=FALSE, verbose=FALSE)

  if (exists("DT.nms.bak")) {
    setkeyIfNot(DT.nms.bak, "nms", organize=TRUE, warnForColNameInEnv=FALSE, verbose=FALSE)
    DT.nms[DT.nms.bak,  ret_json := i.ret_json]
    rm(DT.nms.bak)
  }
}

## Organize the names, because I believe there are API limits
## first the global, then anything with 2014
nms.ordered <- {
  c(
    DT.nms[country == "global" & date >= "2014-02-01", nms],
    DT.nms[country %in% c("US", "SE") & date >= "2014-02-01", nms],
    DT.nms[country %ni% c("global", "US", "SE") & date >= "2014-02-01", nms],
    DT.nms[country %in% c("global", "US", "SE") & date < "2014-02-01", nms],
    DT.nms[country %ni% c("global", "US", "SE") & date < "2014-02-01", nms]
    )
}
stopifnot(identical(sort(nms.ordered), sort(DT.nms$nms)))
## TODO:  Create a DT of scrape_success. 
##        Where results can be either (already present, success, error)

for (nm in nms.ordered) {
## Recovery after error
# for (nm in  nms.ordered[(which(nms.ordered == nm) + 1) : length(nms.ordered)]) {
  cat("Processing : ", nm, "\n")

    ## don't pull again if already pulled
    if (exists("DTs.list") && isTRUE(nrow( DTs.list[[nm]] ) >= nrowThresh)) {
        "do nothing"
    } else {
        ret <- try(getURL(DT.nms[.(nm)]$url))
        if (isErr(ret)) {
          warning (sprintf("ERROR:  '%s',  '%s'", DT.nms[.(nm)]$country, DT.nms[.(nm)]$date))
        } else {
          if (!grepl("^.\"tracks\"", substr(ret, 1, 100)))
            warning (sprintf("POSSIBLE ERROR:  '%s',  '%s'", DT.nms[.(nm)]$country, DT.nms[.(nm)]$date))
          DT.nms[.(nm), ret_json := ret]
        }      
    }
    if (nm %in% nms.ordered[unique(1 + seq_along(nms.ordered) %/% 100) * 100 ])
      jesusForData(DT.nms, info="mid step save")
}

jesusForData(DT.nms, info="with raw data")
f.lastSaved <- saveImageTo()

### -------------------  THIS SHOULD SPLIT INTO SECOND FILE ------------------ ###
##                                                                              ##
##                            EVERYTHING THAT FOLLOWS                           ##
##                            EVERYTHING THAT FOLLOWS                           ##
##                            EVERYTHING THAT FOLLOWS                           ##
##                                                                              ##
### -------------------  THIS SHOULD SPLIT INTO SECOND FILE ------------------ ###
### -------------------  THIS SHOULD SPLIT INTO SECOND FILE ------------------ ###
### -------------------  THIS SHOULD SPLIT INTO SECOND FILE ------------------ ###
### -------------------  THIS SHOULD SPLIT INTO SECOND FILE ------------------ ###

{
  tracks.flat <- unlist(DT.nms[date != Sys.Date() & substr(ret_json, 1, 27) != "<!DOCTYPE html>\n<html lang=", 
                    setNames(nm=nms, obj=ret_json)])
  if (!exists("errors") || is.null(errors)) {errors <- character()}

  ## There is an issue of un-escaped quotes
  tracks.flat <-  gsub('Axwell /. Ingrosso', 'Axwell Ingrosso',  tracks.flat)
  tracks.flat <-  gsub('Nat "King" Cole', 'Nat \\\\"King\\\\" Cole',  tracks.flat)
  tracks.flat <-  gsub('Immortals - From "Big Hero 6.', 'Immortals - From \\\\"BigHero6\\\\"',  tracks.flat)
  tracks.flat <-  gsub('From "The Hunger Games: Mockingjay Part 1" Soundtrack', 'From \\\\"The Hunger Games: Mockingjay Part 1\\\\" Soundtrack',  tracks.flat)
  tracks.flat <-  gsub('From "The Hunger Games: Catching Fire" Soundtrack', 'From \\\\"The Hunger Games: Catching Fire\\\\" Soundtrack',  tracks.flat)
  tracks.flat <-  gsub('"Elastic Heart . From "The Hunger Games. Catching Fire".Soundtrack"', '"Elastic Heart - From \\\\"The Hunger Games: Catching Fire\\\\"/Soundtrack"',  tracks.flat)
  tracks.flat <-  gsub('"Happy . From "Despicable Me 2""', '"Happy - From \\\\"Despicable Me 2\\\\""',  tracks.flat)
  tracks.flat <-  gsub('"Happy .From "Despicable Me 2"."', '"Happy (From \\\\"Despicable Me 2\\\\")"',  tracks.flat)
  tracks.flat <-  gsub('""Riptide""', '"\\\\"Riptide\\\\""',  tracks.flat)
  tracks.flat <-  gsub('""TIME FLIES" EP 2010"', '"\\\\"TIME FLIES\\\\" EP 2010"',  tracks.flat)
  tracks.flat <-  gsub('"Back in Time . featured in "Men In Black 3""', '"Back in Time - featured in \\\\"Men In Black 3\\\\""',  tracks.flat)
  tracks.flat <-  gsub('From "McFarland, USA"', 'From \\\\"McFarland, USA\\\\"',  tracks.flat)
  tracks.flat <-  gsub(' "La Voz""', ' \\\\"La Voz\\\\""',  tracks.flat)
  tracks.flat <-  gsub('Soundtrack "Honig im Kopf"', 'Soundtrack \\\\"Honig im Kopf\\\\"',  tracks.flat)
  tracks.flat <-  gsub('From "Teenage Mutant Ninja Turtles""', 'From \\\\"Teenage Mutant Ninja Turtles\\\\""',  tracks.flat)
  tracks.flat <-  gsub('"Tito "El Bambino" El Patr', '"Tito \\\\"El Bambino\\\\" El Patr',  tracks.flat)
  tracks.flat <-  gsub('"Black Pearl "He\'s A Pirate"', '"Black Pearl \\\\"He\'s A Pirate\\\\"',  tracks.flat)
  tracks.flat <-  gsub('"Never Seen Anything "Quite Like You""', '"Never Seen Anything \\\\"Quite Like You\\\\""',  tracks.flat)

  ## Note two errors with Fifty Shades of Grey -- one with space before the quote, one after
  tracks.flat <-  gsub('" Fifty Shades Of Grey"', '\\\\" Fifty Shades Of Grey\\\\"',  tracks.flat)
  tracks.flat <-  gsub(' "Fifty Shades Of Grey"', ' \\\\"Fifty Shades Of Grey\\\\"',  tracks.flat)


  ## Foreign language Errors
  tracks.flat <-  gsub('Dian Shi Ju "Yi Jian Bu Zh?ong Qing" ', 'Dian Shi Ju \\\\"Yi Jian Bu Zhong Qing\\\\" ', tracks.flat)
  # tracks.flat <-  gsub('Dian Shi Ju "Yi Jian Bu Zhong Qing" Cha Qu', 'Dian Shi Ju \\\\"Yi Jian Bu Zhong Qing\\\\" Cha Qu', tracks.flat)
  # tracks.flat <-  gsub('Dian Shi Ju "Yi Jian Bu Zong Qing" Pian Wei', 'Dian Shi Ju \\\\"Yi Jian Bu Zong Qing\\\\" Pian Wei', tracks.flat)
  tracks.flat <-  gsub('劇"一見不鍾情"', '劇\\\\"一見不鍾情\\\\"', tracks.flat)
  # tracks.flat <-  gsub('劇"一見不鍾情" 插', '劇\\\\"一見不鍾情\\\\" 插', tracks.flat)
  # tracks.flat <-  gsub('劇"一見不鍾情"片',  '劇\\\\"一見不鍾情\\\\"片',  tracks.flat)
}

if (!exists("DTs.list")) {
  DTs.list <- emptylist(tracks.flat)
}


verboseMsg(verbose, "Beginning Parsing", time=TRUE)
for (tx.nm in names(tracks.flat[!is.na(tracks.flat)])) {
  cat("\n  Parsing ", tx.nm, "")
  tx <- tracks.flat[[tx.nm]]
  txl <- try( fromJSON(tx)[["tracks"]] )
  if (isErr(txl))  {
    warning ("ERROR: ", tx.nm)
    errors <- c(errors, tx.nm)
  } else
    DTs.list[[tx.nm]] <-  do.call(rbind, lapply(txl, function(x) {x[sapply(x, is.null)] <- ""; as.data.table(x)} ))
}
errors <- unique(errors)

verboseMsg(verbose, "\nDone Parsing", time=TRUE)


jesusForData(DTs.list, info=ifelse(length(errors), "has blanks from errors", ""))
notifyAndEmail("Spotify Charts API Complete")

# error fixing #  -   ------------------------------------------------------------------------
" Use this to manually check for errors"
if (FALSE)
  if (length(errors)) {
   
    if (FALSE && "this is the Manual part") 
    {
       ## Put in the name of the culprit here
       tx.nm <- "MY.2014-11-23"
       tx.nm <- "US.2014-12-14"
       tx.nm <- "AT.2015-01-11"
       tx.nm <- tail(errors, 1)


       tx <- tracks.flat[[tx.nm]]
       .cc(tx)
       system("open http://pro.jsonlint.com")
       
       ## If found, we need to manually create a gsub for it.  
       ## First we need to find the exact way(s) it appears
       ## (step 1): Set the keyword 
       ## (step 2): Manually examine the output. Explore the bottom row for the improperly escaped offending character. The copy the top row for the gsub
       ## (step 3): Paste the toprow as a gsub, but it still needs more cleaning for the search pattern and for the replace. 
       ## (step 4): run the gsub
       ## (step 5): Run the next section
       ## 
       "-------------------------------------------"
       "use this part, manually to see the original"
       "-------------------------------------------"
       keyword <- "Fifty Shades Of Grey"
       keyword <- "Love Me Like You Do"
       keyword <- 'Go Solo \\(From the Original'
       {iii <- gregexpr(keyword, tx)[[1]]; if (all(iii == -1)) warning("no match for '", keyword, "' in  ", tx.nm) else for (ii in iii) {
        ss <-substr(tx, ii-40, ii+100); cat("\n"); print(ss); cat("    ", ss, "\n\n")}}
        ## CREATE the pattern, then test it 
        {
          test_string <- "6Tp\",\"track_name\":\"我想愛(電視劇\"一見不鍾情\" 插曲) - Dian Shi Ju \"Yi Jian Bu Zhong Qing\" Cha Qu\",\"artist_name\":\"楊丞琳\",\"artist_url\":\"https://play.spotify.co"
          pat  <- 'Dian Shi Ju \"Yi Jian Bu Zhong Qing\" Pian Wei'
          pat  <- 'Dian Shi Ju .Yi Jian Bu Zhong Qing. Pian Wei'
          repl <- 'Dian Shi Ju \\\\"Yi Jian Bu Zhong Qing\\\\" Pian Wei'
          repl <- paste("{-@", repl, "@-}")
          result_string <- gsub(pat, repl, test_string)
          if (!grepl(pat, test_string)) {
            message ("pattern not found in test_string")
            print (pat)
            print(test_string)
          } else 
            cat(pasteR(40), "\n", test_string, "\n\n",result_string, "\n", pasteR(40), "\n")
        }
    }

    ## RUN THIS AFTER RE-RUNNING THE GSUB 
    if (TRUE) {
        if (length(errors)) {
          errors_fixing <- unique(errors)
          errors <- c()
          cat ("   ", length(errors_fixing), "errors remain\n")
        }
        for (tx.nm in errors_fixing) {
            tx <- tracks.flat[[tx.nm]]
            txl <- try( fromJSON(tx)[["tracks"]], silent=FALSE )
            if (isErr(txl))  {
              warning ("ERROR: ", tx.nm)
              errors <- c(errors, tx.nm)
            } else {
              message ("RESOLVED: ", tx.nm)
              DTs.list[[tx.nm]] <-  do.call(rbind, lapply(txl, function(x) {
                x[sapply(x, is.null)] <- ""; as.data.table(x)} ))
            }
        }
    } # // END RE-RUNNING AFTER GSUB
  }
# error fixing #  -   ------------------------------------------------------------------------

## If no errors, let user know
verboseMsg(!length(errors), "Good job! No Errors in the JSON parsing", func="message")

## DROP THE groups WITH NO ROWS
DTs.list[sapply(DTs.list, is.null)] <- NULL

rows <- sapply(DTs.list, nrow)
rows[sapply(rows, is.null)] <- 0
rows <- unlist(rows)
## It looks like Spotify only includes tracks with num_streams >= 1001
# DTs.list[rows < nrowThresh] <- NULL
cat("These would other wise be dropped")
print(rbindlist(lapply(which(rows < 10 & !sapply(DTs.list, is.null)), function(i) cbind(DTs.list[[i]] [1, list(country, date)], rows=rows[[i]] ) )) [, list(dates_per_country = .N), by=country][order(dates_per_country, decreasing=TRUE)])


## SEE
rows[rows < 50]
rows[rows == 1]

stopifnot(all(sapply(DTs.list, ncol) == ncol(DTs.list[[1]])))
DT.toptracks <- rbindlist(DTs.list)
setInfo(DT.toptracks, "rbindlist of DTs.list, which itself is a collection of data.tables of the JSON parse of the track info from the Spotify Web Scrape, by country and week")

jesusForData(tracks.flat)
jesusForData(DT.nms)
jesusForData(DT.toptracks)
jesusForData(DTs.list)


# .g()
# loadFromJesus("tracks.flat", over=TRUE)
# loadFromJesus("DT.nms", over=TRUE)
# loadFromJesus("DT.toptracks", over=TRUE)
# loadFromJesus("DTs.list", over=TRUE)





