#!/bin/sh
# sfeed_download: downloader for URLs and enclosures in sfeed(5) files.
# Dependencies: awk, curl, flock, xargs (-P), yt-dlp.

cachefile="${SFEED_CACHEFILE:-$HOME/.sfeed/downloaded_urls}"
jobs="${SFEED_JOBS:-4}"
lockfile="${HOME}/.sfeed/sfeed_download.lock"

# log(feedname, s, status)
log() {
	if [ "$1" != "-" ]; then
		s="[$1] $2"
	else
		s="$2"
	fi
	printf '[%s]: %s: %s\n' "$(date +'%H:%M:%S')" "${s}" "$3"
}

# fetch(url, feedname)
fetch() {
	case "$1" in
	*youtube.com*)
		yt-dlp "$1";;
	*.flac|*.ogg|*.m3u|*.m3u8|*.m4a|*.mkv|*.mp3|*.mp4|*.wav|*.webm)
		# allow 2 redirects, hide User-Agent, connect timeout is 15 seconds.
		curl -O -L --max-redirs 2 -H "User-Agent:" -f -s --connect-timeout 15 "$1";;
	esac
}

# downloader(url, title, feedname)
downloader() {
	url="$1"
	title="$2"
	feedname="${3##*/}"

	msg="${title}: ${url}"

	# download directory.
	if [ "${feedname}" != "-" ]; then
		mkdir -p "${feedname}"
		if ! cd "${feedname}"; then
			log "${feedname}" "${msg}: ${feedname}" "DIR FAIL" >&2
			return 1
		fi
	fi

	log "${feedname}" "${msg}" "START"
	if fetch "${url}" "${feedname}"; then
		log "${feedname}" "${msg}" "OK"

		# append it safely in parallel to the cachefile on a
		# successful download.
		(flock 9 || exit 1
		printf '%s\n' "${url}" >> "${cachefile}"
		) 9>"${lockfile}"
	else
		log "${feedname}" "${msg}" "FAIL" >&2
		return 1
	fi
	return 0
}

if [ "${SFEED_DOWNLOAD_CHILD}" = "1" ]; then
	# Downloader helper for parallel downloading.
	# Receives arguments: $1 = URL, $2 = title, $3 = feed filename or "-".
	# It should write the URI to the cachefile if it is successful.
	downloader "$1" "$2" "$3"
	exit $?
fi

# ...else parent mode:

tmp="$(mktemp)" || exit 1
trap "rm -f ${tmp}" EXIT

[ -f "${cachefile}" ] || touch "${cachefile}"
cat "${cachefile}" > "${tmp}"
echo >> "${tmp}" # force it to have one line for awk.

LC_ALL=C awk -F '\t' '
# fast prefilter what to download or not.
function filter(url, field, feedname) {
	u = tolower(url);
	return (match(u, "youtube\\.com") ||
	        match(u, "\\.(flac|ogg|m3u|m3u8|m4a|mkv|mp3|mp4|wav|webm)$"));
}
function download(url, field, title, filename) {
	if (!length(url) || urls[url] || !filter(url, field, filename))
		return;
	# NUL-separated for xargs -0.
	printf("%s%c%s%c%s%c", url, 0, title, 0, filename, 0);
	urls[url] = 1; # print once
}
{
	FILENR += (FNR == 1);
}
# lookup table from cachefile which contains downloaded URLs.
FILENR == 1 {
	urls[$0] = 1;
}
# feed file(s).
FILENR != 1 {
	download($3, 3, $2, FILENAME); # link
	download($8, 8, $2, FILENAME); # enclosure
}
' "${tmp}" "${@:--}" | \
SFEED_DOWNLOAD_CHILD="1" xargs -r -0 -L 3 -P "${jobs}" "$(readlink -f "$0")"
