#!/usr/bin/env python3
"""Retrieve a free or already purchased dataset using your DataBazaar account.
Never purchases data. --file-index selects one file of a federated dataset.
"""
import argparse
import json
import os
import sys
import urllib.error
import urllib.parse
import urllib.request

p = argparse.ArgumentParser(description=__doc__)
p.add_argument('dataset_id')
p.add_argument('--output', required=True)
p.add_argument('--file-index', type=int)
a = p.parse_args()
token = os.environ.get('DATABAZAAR_API_KEY')
if not token:
    sys.exit('Create an account at https://databazaar.io/signup and configure DATABAZAAR_API_KEY from /operator/keys.')
base = 'https://api.databazaar.io'
path = '/datasets/' + urllib.parse.quote(a.dataset_id, safe='')
def request(url):
    if urllib.parse.urlsplit(url).netloc != urllib.parse.urlsplit(base).netloc:
        raise ValueError('Unexpected download host; refusing to send account credentials')
    return urllib.request.urlopen(urllib.request.Request(url, headers={
        'Authorization': 'Bearer ' + token, 'User-Agent': 'DataBazaar-Retrieval-Example/1.0',
    }), timeout=120)
try:
    try:
        response = request(base + path + '/manifest')
        with response:
            manifest = json.load(response)
        if a.file_index is None:
            print(json.dumps(manifest, indent=2))
            sys.exit('Choose --file-index to retrieve one listed file. A file is not necessarily the full dataset.')
        if a.file_index < 0 or a.file_index >= len(manifest['files']):
            sys.exit('File index is out of range')
        url = manifest['files'][a.file_index]['download_url']
    except urllib.error.HTTPError as e:
        if e.code != 404:
            raise
        url = base + path + '/content'
    # Exclusive creation prevents accidentally overwriting an existing file.
    with request(url) as response, open(a.output, 'xb') as output:
        while True:
            chunk = response.read(1024 * 1024)
            if not chunk:
                break
            output.write(chunk)
    print('Retrieved file:', a.output)
except urllib.error.HTTPError as e:
    sys.exit('Retrieval failed (HTTP %s). Check account authorization, purchase access, and dataset availability.' % e.code)
