diff options
Diffstat (limited to 'reproduce/analysis/bash')
| -rwxr-xr-x | reproduce/analysis/bash/download-multi-try.sh (renamed from reproduce/analysis/bash/download-multi-try) | 100 |
1 files changed, 76 insertions, 24 deletions
diff --git a/reproduce/analysis/bash/download-multi-try b/reproduce/analysis/bash/download-multi-try.sh index 994a8fa..2df8aab 100755 --- a/reproduce/analysis/bash/download-multi-try +++ b/reproduce/analysis/bash/download-multi-try.sh @@ -1,14 +1,19 @@ -#!/bin/sh +#!/usr/bin/env sh # # Attempt downloading multiple times before crashing whole project. From # the top project directory (for the shebang above), this script must be # run like this: # -# $ /path/to/download-multi-try downloader lockfile input-url downloaded-name +# $ $SHELL /path/to/download-multi-try.sh downloader lockfile \ +# input-url downloaded-name # -# NOTE: The 'downloader' must contain the option to specify the output name -# in its end. For example "wget -O". Any other option can also be placed in -# the middle. +# NOTE: +# - This script doesn't have a Shebang because in different stages it +# should be built with different shells ('/bin/sh' before Maneage +# installs its own shell and afterwards with Maneage's own shell). +# - The 'downloader' must contain the option to specify the output name +# in its end. For example "wget -O". Any other option can also be placed in +# the middle. # # Due to temporary network problems, a download may fail suddenly, but # succeed in a second try a few seconds later. Without this script that @@ -26,7 +31,7 @@ # reason, you don't want to use a lock file, set the 'lockfile' name to # 'nolock'. # -# Copyright (C) 2019-2022 Mohammad Akhlaghi <mohammad@akhlaghi.org> +# Copyright (C) 2019-2026 Mohammad Akhlaghi <mohammad@akhlaghi.org> # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU General Public License as published by @@ -85,6 +90,61 @@ urlfile=$(echo "$inurl" | awk -F "/" '{print $NF}') +# Function for downloading +download_func () { + + # Set the arguments. + inurl="$1" + + # During the installation of basic software, the dependencies of the + # downloaders that are installed in Maneage can conflict with the + # downloader that is not yet installed in Maneage. To avoid this, we + # check if the downloader works (with '--version') and remove the + # Maneage library temporarily in case it does not. + lpathorig="" + dprog=$(echo $downloader | awk '{print $1}') + if ! $dprog --version > /dev/null 2> /dev/null; then + echo "NOTE: the segmentation fault is expected, is not a problem" + basedir=$(echo $outname \ + | sed -e's|\/tarballs\/| |' \ + | awk '{print $1}') + libdir=$basedir/installed/lib + lpathorig=$LD_LIBRARY_PATH + lpath=$(echo $LD_LIBRARY_PATH \ + | sed -e's|'$libdir':||') + export LD_LIBRARY_PATH=$lpath + fi + + # Attempt downloading the file. Note that the 'downloader' ends with + # the respective option to specify the output name. For example "wget + # -O" (so 'outname', that comes after it) will be the name of the + # downloaded file. + if [ x"$lockfile" = xnolock ]; then + if ! $downloader $outname $inurl; then rm -f $outname; fi + else + flock "$lockfile" sh -c \ + "if ! $downloader $outname \"$inurl\"; then rm -f $outname; fi" + fi + + # In case the LD_LIBRARY_PATH was changed, set it back to normal. + if ! [ x"$lpathorig" = x ]; then + export LD_LIBRARY_PATH=$lpathorig + fi + + # Some servers return HTTP 4xx/5xx errors as an HTML page with a 200 + # status, so the downloader exits 0 and saves the HTML body to disk. + # Detect this by checking whether the response starts with an HTML tag + # and treat it as a failed download so backup servers will be tried. + if [ -f "$outname" ] \ + && head -c 500 "$outname" 2>/dev/null | grep -qi '<html'; then + rm -f "$outname" + fi +} + + + + + # Try downloading multiple times before crashing. counter=0 maxcounter=10 @@ -114,30 +174,22 @@ while [ ! -f "$outname" ]; do sleep $tstep fi - # Attempt downloading the file. Note that the 'downloader' ends with - # the respective option to specify the output name. For example "wget - # -O" (so 'outname', that comes after it) will be the name of the - # downloaded file. - if [ x"$lockfile" = xnolock ]; then - if ! $downloader $outname $inurl; then rm -f $outname; fi - else - # Try downloading from the requested URL. - flock "$lockfile" sh -c \ - "if ! $downloader $outname $inurl; then rm -f $outname; fi" - fi + # First attempt at the download. + download_func "$inurl" # If the download failed, try the backup server(s). if [ ! -f "$outname" ]; then if [ x"$backupservers" != x ]; then for bs in $backupservers; do - # Use this backup server. - if [ x"$lockfile" = xnolock ]; then - if ! $downloader $outname $bs/$urlfile; then rm -f $outname; fi - else - flock "$lockfile" sh -c \ - "if ! $downloader $outname $bs/$urlfile; then rm -f $outname; fi" - fi + # For Zenodo backup servers, append '?download=1' as well. + bsurl="$bs/$urlfile" + case "$bsurl" in + *zenodo.org*) bsurl="${bsurl}?download=1" ;; + esac + + # Download the file. + download_func "$bsurl" # If the file was downloaded, break out of the loop that # parses over the backup servers. |
