#!/usr/bin/env python3
"""Inspect a public sample, or export an account-authenticated query to CSV.
Python 3.10+, standard library only. Never purchases data. No credentials in files.
"""
import argparse
import collections
import csv
import hashlib
import io
import json
import os
import sys
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path

API = 'https://api.databazaar.io'
WORKFLOWS = {
    'deepsearchqa': ('24cf80b3-5f47-4da6-a6a5-0540b3688343', ['problem', 'problem_category', 'answer', 'answer_type']),
    'stock-bars': ('89d911b1-31ee-40eb-bd09-fc3ef75a5838', ['timestamp', 'ticker', 'open', 'high', 'low', 'close', 'volume']),
}
MAX_BYTES = 8 * 1024 * 1024

class SameOriginRedirect(urllib.request.HTTPRedirectHandler):
    def redirect_request(self, req, fp, code, msg, headers, newurl):
        if req.has_header('Authorization') and urllib.parse.urlsplit(newurl).netloc != urllib.parse.urlsplit(API).netloc:
            raise ValueError('Refusing to forward account credentials to another host')
        if urllib.parse.urlsplit(newurl).scheme != 'https':
            raise ValueError('HTTPS is required')
        return super().redirect_request(req, fp, code, msg, headers, newurl)

def request(url, key=None, body=None, synthetic=False):
    parsed = urllib.parse.urlsplit(url)
    if parsed.scheme != 'https' or parsed.username or parsed.password:
        raise ValueError('Only credential-free HTTPS URLs are supported')
    if key and parsed.netloc != urllib.parse.urlsplit(API).netloc:
        raise ValueError('Account credentials may only be sent to the DataBazaar API')
    headers = {'User-Agent': 'DataBazaar-workflow' + ('-synthetic-check' if synthetic else '')}
    if key: headers['Authorization'] = 'Bearer ' + key
    if body is not None: headers['Content-Type'] = 'application/json'
    req = urllib.request.Request(url, data=json.dumps(body).encode() if body is not None else None, headers=headers)
    with urllib.request.build_opener(SameOriginRedirect()).open(req, timeout=60) as res:
        raw = res.read(MAX_BYTES + 1)
        if len(raw) > MAX_BYTES: raise ValueError('Response exceeds the example size limit')
        return raw

def validate_rows(rows, columns):
    if not isinstance(rows, list) or not rows or any(not isinstance(r, dict) or not set(columns).issubset(r) for r in rows):
        raise ValueError('No matching records returned; the listing or query schema may have changed')
    return rows

def parse_sample(raw, sample_format):
    text = raw.decode('utf-8-sig')
    if sample_format == 'json':
        return json.loads(text)
    if sample_format == 'csv':
        reader = csv.reader(io.StringIO(text, newline=''), strict=True)
        header = next(reader, [])
        if not header or any(not name for name in header) or len(set(header)) != len(header):
            raise ValueError('CSV sample has missing or duplicate column names')
        rows = []
        for values in reader:
            if len(values) != len(header):
                raise ValueError('CSV sample row does not match its header')
            rows.append(dict(zip(header, values)))
        return rows
    raise ValueError(f'Unsupported public sample format: {sample_format!r}')

def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('workflow', choices=WORKFLOWS)
    parser.add_argument('--authenticated', action='store_true', help='Query real data using DATABAZAAR_API_KEY; does not purchase')
    parser.add_argument('--limit', type=int, default=1000)
    parser.add_argument('--output', type=Path, help='New CSV path; existing files are never overwritten')
    parser.add_argument('--synthetic-check', action='store_true', help='Exclude maintenance checks from adoption metrics')
    args = parser.parse_args()
    if not 1 <= args.limit <= 1000: parser.error('--limit must be between 1 and 1000')
    if args.output and not args.authenticated: parser.error('--output requires --authenticated; public mode only inspects a sample')
    if args.output and args.output.exists(): parser.error('Output file already exists')
    key = os.environ.get('DATABAZAAR_API_KEY') if args.authenticated else None
    if args.authenticated and not key: parser.error('Set DATABAZAAR_API_KEY after creating your account at https://databazaar.io/operator/keys')
    dataset, columns = WORKFLOWS[args.workflow]
    if args.authenticated:
        raw = request(f'{API}/datasets/{dataset}/query', key=key, body={'select': columns, 'limit': args.limit}, synthetic=args.synthetic_check)
        rows = [json.loads(line) for line in raw.decode('utf-8').splitlines() if line.strip()]
    else:
        info = json.loads(request(f'{API}/datasets/{dataset}/sample', synthetic=args.synthetic_check))
        raw = request(info['sample_url'], synthetic=args.synthetic_check)  # No account key sent to storage.
        rows = parse_sample(raw, info.get('sample_format'))
    rows = validate_rows(rows, columns)
    if args.output:
        with args.output.open('x', encoding='utf-8', newline='') as f:
            writer = csv.DictWriter(f, fieldnames=columns, extrasaction='ignore')
            writer.writeheader()
            writer.writerows(rows)
    summary = {'success': True, 'dataset_id': dataset, 'scope': 'authenticated_query' if args.authenticated else 'sample_only', 'rows': len(rows), 'columns': columns, 'response_sha256': hashlib.sha256(raw).hexdigest()}
    if args.authenticated:
        summary['requested_limit'] = args.limit
        summary['complete_dataset_verified'] = False
    if args.workflow == 'deepsearchqa':
        summary['category_counts_in_returned_rows'] = dict(collections.Counter(str(r['problem_category']) for r in rows))
        summary['next_step'] = 'Send only problem to your agent; keep answer for evaluation. Use the source benchmark evaluation protocol.'
    else:
        summary['tickers_in_returned_rows'] = sorted({str(r['ticker']) for r in rows})
        summary['next_step'] = 'Validate timestamps, gaps, corporate actions and license terms before broader analysis. This is an unordered bounded slice, not a market-wide summary.'
    if args.output: summary['output'] = str(args.output)
    print(json.dumps(summary, indent=2))

if __name__ == '__main__':
    try: main()
    except urllib.error.HTTPError as exc:
        print(json.dumps({'success': False, 'error': f'HTTP {exc.code}. Check account access, listing availability and query support.'}), file=sys.stderr); sys.exit(1)
    except (ValueError, OSError, csv.Error) as exc:
        print(json.dumps({'success': False, 'error': str(exc)}), file=sys.stderr); sys.exit(1)
