#!/usr/bin/env bash

bashColors=".bash_colors"
dlBashColors="https://raw.githubusercontent.com/mercuriev/bash_colors/master/bash_colors.sh"
wgetCol="collections/wget"
heritrixCol="collections/heritrix"
wgetUA="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/61.0.3163.100 Safari/537.36"
wgetArgs="--mirror
--warc-cdx
--no-warc-compression
--adjust-extension
--page-requisites
--html-extension
--convert-links
--execute robots=off
--no-directories
--span-hosts
"

function curlExists () {
    type -t "curl" | grep -q 'function'
}

function pvExists () {
  type -t "pv" | grep -q 'function'
}

if [[ -f "$bashColors" ]]; then
  source ${bashColors}
else
  if [[ curlExists ]]; then
    curl ${dlBashColors} > ${bashColors}
  else
    wget --user-agent="$wgetUA" -O "$dlBashColors" "$bashColors"
  fi
  source ${bashColors}
  if [[ ! curlExists ]]; then
    echo "$(clr_bold clr_red You Need To Download Curl)"
  fi
fi

function setupWayback () {
  if [[ ! -d "$wgetCol" ]]; then
    wb-manager init wget
  fi
  if [[ ! -d "$heritrixCol" ]]; then
    wb-manager init heritrix
  fi
}

function checkDirs () {
  for maybeDir in "$@"
  do
    if [[ ! -d $maybeDir ]]; then
      mkdir ${maybeDir}
    fi
  done
}

function wgetArchive () {
  if [[ ! -f "$1.warc" ]]; then
    if [[ $2 = "http://example.com" ]]; then
      wget ${wgetArgs} --user-agent="$wgetUA" --domains=example.com,www.example.com,cdn.example.com \
      --warc-file="$1" "$2"
    else
      wget ${wgetArgs} --user-agent="$wgetUA"  --warc-file="$1" "$2"
    fi
  fi
}

function copyWarcCdxTo () {
  cp *.cdx *.warc "$1"
}



function wgetIntro() {
  echo "$(clr_bold clr_cyan Hello) I will be using $(clr_bold clr_cyan wget) to archive two web pages" | pv -qL 25
  echo "The arguments you give to wget are:" | pv -qL 25
  for it in $wgetArgs
  do
    argColor=$(clr_escape "$it" $CLR_BOLD $CLR_CYAN)
    case "$it" in
      "--mirror" )
        echo -e "${argColor}: Recursive infinite crawl depth with timesamping" | pv -qL 25
      ;;
      "--warc-cdx")
        boo=$(clr_escape "--warc-file" $CLR_BOLD $CLR_CYAN)
        echo -e "${boo}: Create the WARC file with the supplied name" | pv -qL 25
        echo -e "${argColor}: Create a cdx file along side the WARC file" | pv -qL 25
      ;;
      "--no-warc-compression")
        echo -e "${argColor}: Do not gzip compress the WARC file" | pv -qL 25
      ;;
      "--adjust-extension")
        echo -e "${argColor}: Ensure files with MIME type ‘application/xhtml+xml’ or ‘text/html’ files are saved as .html" | pv -qL 25
      ;;
      "--page-requisites")
        echo -e "${argColor}: Download all the files 'necessary' to view the given HTML page" | pv -qL 25
      ;;
      "--convert-links")
        echo -e "${argColor}: Make the links contained in the downloaded html files relative to the directory they were saved in" | pv -qL 25
      ;;
      "robots=off")
        boo=$(clr_escape "--execute" $CLR_BOLD $CLR_CYAN)
        echo -e "${boo} ${argColor}: Disregard the robots.txt file" | pv -qL 25
      ;;
      "--no-directories")
        echo -e "${argColor}: Do not create a directory structure that matches the path of the URL" | pv -qL 25
      ;;
      "--span-hosts")
        echo -e ${argColor}: "Crawl all the hosts" | pv -qL 25
      ;;
    esac
  done
  argColor=$(clr_escape "--user-agent" $CLR_BOLD $CLR_CYAN)
  echo -e ${argColor}: "Set the user-agent string for wget" | pv -qL 25
}
