"""
Build an RSS document from registry info.
# TODO: throw out unnecessary items using res_type
"""

import datetime
import os
import re
import random
import string
import sys
import time
import traceback

from mastodon import Mastodon

from gavo import api
from gavo import base
from gavo import utils # helpers and wrappers for dachs' python modules
from gavo import votable
from gavo.utils import stanxml # stan-like model for building namespaced xml trees
from gavo.votable import tapquery

RD = api.getRD("registryrss/q")

# base access URL of the TAP service containing the registry data
TAP_URL = "http://dc.g-vo.org/tap"
# number of days to show
N_DAYS = 10
# don't include more records in RSS than...
MAXRECS = 40


AUTHORITY_RE = re.compile("ivo://([^/]*)") # compile reg. expression pattern into
# reg. expr. object, which can be used for matching using its e.g. match() or search()
SQL_LIST_SEPARATOR = '=$makerss$='


def getAuthority(ivoid):
	"""returns the authority from an ivoid.

	This raises an AttributeError if ivoid is not an IVORN, but that's an
	implementation detail you shouldn't rely upon.
	"""
	return AUTHORITY_RE.match(ivoid).group(1)


class Ignore(Exception):
	"""raised when a record is unusable in some way that we deem
	permanent.
	"""

# stan-like model for building namespaced XML trees
stanxml.registerPrefix("atom", "http://www.w3.org/2005/Atom", None)

class Atom(object):
	"""the namespace of atom elements.
	"""
	class AtomElement(stanxml.Element):
		_prefix = "atom"

	class AtomTextElement(AtomElement):
		_a_type = None

	class AtomEmptyElement(AtomElement):
		_mayBeEmpty = True

	class author(AtomElement): pass
	class content(AtomTextElement): pass
	class contributor(AtomElement): pass
	class email(AtomElement): pass
	class entry(AtomElement): pass
	class feed(AtomElement): pass
	class generator(AtomElement): pass
	class icon(AtomElement): pass
	class id(AtomElement): pass
	class logo(AtomElement): pass
	class name(AtomElement): pass
	class published(AtomElement): pass
	class rights(AtomElement): pass
	class source(AtomElement): pass
	class subtitle(AtomTextElement):pass
	class summary(AtomTextElement):pass
	class title(AtomTextElement):pass
	class updated(AtomElement): pass
	class uri(AtomElement): pass

	class category(AtomEmptyElement):
		_a_term = None
		_a_scheme = None
		_a_label = None

	class link(AtomEmptyElement):
		_a_href = None
		_a_rel = None
		_a_title = None
		_a_type = None
	

def makeItem(rec):
	# HTML is embedded as plain text in the description element.	Thus,
	# we do daring text operations...
	if "\n" in rec["description"].strip():
		descriptionTemplate = "<pre>%s</pre>"
	else:
		descriptionTemplate = "%s"

	description = ["<dl>",
		"<dt>Description</dt>",
		"<dd>%s</dd>"%(
			descriptionTemplate%stanxml.escapePCDATA(rec["description"]))]
	if rec.get("creator_name"):
		description.extend([
			"<dt>Author(s)</dt>",
			"<dd>%s</dd>"%stanxml.escapePCDATA(rec["creator_name"])])
	description.extend([
		"<dt>IVOA id</dt>",
		"<dd>%s</dd>"%rec["ivoid"]])

	# VizieR has nice footprints -- add those as applicable, using
	# a custom vizier interface based on their catalog names
# (doesn't work right now because of case sensitivity issues, and probably
# others on top)
#	if rec["ivoid"].startswith("ivo://cds.vizier"):
#		catname = rec["ivoid"][17:].upper()
#		description.extend([
#			"<dt>VizieR Footprint</dt>",
#			"<dd><img src='http://alasky.u-strasbg.fr/footprints/cats/vizier/%s"
#			"?product=densityMap&amp;format=png&amp;size=small' alt='[image]'/>"
#			"</dd>"%(catname.upper())])

	description.append("</dl>")

	return Atom.entry[
		Atom.title[rec.get("res_title")],
		Atom.link(rel="alternate", type="text/html", href=rec["reference_url"],
			title="Reference URL"),
		Atom.link(rel="related", href=rec["access_url"], title="Access URL"),
		Atom.id[rec["ivoid"]],
		Atom.updated[rec["updated"].isoformat()+"Z"], [
			Atom.author[
				Atom.name[name]] for name in rec["creator_name"].split(";")],
		Atom.content(type="html")["\n".join(description)], [
			Atom.category(term=kw) for kw in rec["res_subject"]]]


def makeRSS(recs):
	"""returns a stanxml tree of an Atom feed for the registry records in recs.

	recs is a dictionary containing (not necessarily all of) the keys
	ivoid, res_title, reference_url, updated, res_subject (a list), description,
	creator_name (a list)
	"""
	return Atom.feed[
		Atom.title["VO Fresh"],
		Atom.subtitle["New services and resources in the Virtual Observatory,"
				"	as viewed from GAVO's relational registry."],
		Atom.updated[datetime.datetime.utcnow().isoformat()+"Z"],
		Atom.id["ivo://org.gavo.dc/registryrss/q/rss"],
		Atom.link(rel="self", type="application/atom+xml",
			href="http://dc.g-vo.org/regrss"),
		Atom.link(rel="related", type="text/html",
			href="http://www.ivoa.net"),
		Atom.link(rel="related", type="text/html",
			href="http://www.g-vo.org"),
		Atom.author[
			Atom.name["The GAVO data center team"],
			Atom.uri["http://dc.g-vo.org"],
			Atom.email["gavo@ari.uni-heidelberg.de"]],
		Atom.icon["http://vo.uni-hd.de/registryrss/q/rss/static/logo.png"],
		Atom.generator["GAVO DaCHS, makerss module"], [
			makeItem(rec) for rec in recs[:MAXRECS]]]


def preprocessRows(inRows):
	"""joins subject fields from the rows belonging to one RR.
	"""
	for r in inRows:
		if isinstance(r["updated"], str):
			r["updated"] = utils.parseISODT(r["updated"])
		if r["res_subject"] is None:
			r["res_subject"] = []
		else:
			r["res_subject"] = r["res_subject"].split(SQL_LIST_SEPARATOR)

	return sorted(inRows, key=lambda r: r["updated"], reverse=True)


def filterRows(rows):
	"""removes from rows whatever has been tweeted out more than 20 days
	ago.

	This will keep updates to old records off the RSS; perhaps we should
	at some point look at whether there have been "relevant" update.  But
	of course that's not well-defined...
	"""
	newRows = []
	cutoff = datetime.datetime.utcnow()-datetime.timedelta(days=20)

	with base.getTableConn() as conn:
		idTable = api.TableForDef(
			api.resolveCrossId("registryrss/q#sentOut"),
			connection=conn)
		for row in rows:
			try:
				srow = idTable.getRow(row["ivoid"])
				if srow["sent"]>cutoff:
					newRows.append(row)

			except KeyError:
				# not yet sent out
				newRows.append(row)

	return newRows


def getRecsFromTAP():
	"""returns a sequence of registry records suitable for makeRSS
	from a TAP query to TAP_URL.
	"""
	showFrom = datetime.datetime.now()-datetime.timedelta(days=N_DAYS)
	now = datetime.datetime.now()
	job = tapquery.ADQLSyncJob(TAP_URL, """
		SELECT
			ivoid, res_title, reference_url, res_description as description,
			updated, res_subject, creator_seq as creator_name, access_url
 		FROM rr.resource
	 		LEFT OUTER JOIN (
	 	 		SELECT ivoid, ivo_string_agg(res_subject, '%s') as res_subject
	 	 		FROM rr.res_subject
	 	 		GROUP BY ivoid) AS q
 			USING (ivoid)
	 		NATURAL LEFT OUTER JOIN rr.interface
 		WHERE updated between '%s' AND '%s'
	 	 	 AND cap_index=1
	 	 	 AND intf_index=1
 		ORDER BY updated DESC
	"""%(SQL_LIST_SEPARATOR, showFrom.isoformat(), now.isoformat()))
	job.run()
	data, metadata = votable.load(job.openResult())

	rows = preprocessRows(list(metadata.iterDicts(data)))
	rows = filterRows(rows)
	return rows


############################# begin microblogging interface
MIN_MICROBLOG_PERIOD = 60000


def getCredentials(fName):
	"""returns a dictionary of credential parts to their values.

	This must return a dictionary of keyword arguments for the Mastdon
	constructor.

	For python3-mastodon, that's client_secret, access_token, api_base_url.
	"""
	res = {}
	with open(RD.getAbsPath(fName)) as f:
		for ln in f:
			key, value = ln.split()
			res[key] = value
	return res


def subjectToHashtag(subject):
	"""returns something hopefully suitable for a hashtag from
	registry subjects.

	That's likely ok with UAT words, possibly crap with freetext.
	"""
	return string.capwords(subject.replace("-", " ")).replace(" ", "")

	
def getMicroblogText(rec):
	"""returns a dictionary containing items for representing the
	resource rec in a microblogging service.
	"""
	theURL = rec["reference_url"]
	if (not theURL
			or not theURL.startswith("http")):
		raise Ignore("Silly URL")
	
	headline = rec["res_title"]

	if rec["creator_name"]:
		names = rec["creator_name"].split(";")
		if len(names) >= 2:
			authors = " by "+names[0] + " et al."
		else:
			authors = " by "+names[0]
	else:
		authors = ""

	hashtags = " ".join("#"+subjectToHashtag(subject)
		for subject in rec["res_subject"][:4] if subject.strip())
	if hashtags:
		hashtags = "\n"+hashtags

	return (f"New in the #VirtualObservatory: “{headline}”{authors}\n"
		f"{theURL}{hashtags}")


def sendOneToMicroblog(rec, idTable, now, mastodon_service):
	ivo_id = rec["ivoid"]
	try:
		lastPublished = idTable.getRow(ivo_id)["sent"]
	except KeyError:	# new ivo id, not yet in DB, so not yet published
		lastPublished = None
	if (lastPublished is not None
			and lastPublished>now-datetime.timedelta(days=MIN_MICROBLOG_PERIOD)):
		return # if already published (since entry in table exists) and within
				# the blogging period, return
	try:
		text = getMicroblogText(rec)
	except KeyError:	# incomplete record, ignore
		return
	except Ignore: # permanently broken record, don't send but ignore
		text = None

	try:
		if text is not None:
			time.sleep(random.randint(90, 500))
			if len(text)<=1000:
				mastodon_service.toot(text)
			else:
				from gavo.base import cron
				cron.sendMailToAdmin(
					"registryrss: overlong status: %s"%repr(text))
	except (IOError):
		raise
	# Add the ivoid of the sent toot to sentOut table
	idTable.addRow({"ivo_id": ivo_id, "sent": now})
	idTable.connection.commit()


def sendToMicroblog(recs):

	now = datetime.datetime.utcnow()
	rd = api.getRD("registryrss/q")
	# getCredentials assumes oauth2 keys.
	mastodon_service = Mastodon(**getCredentials("fed_creds.txt"))

	sentOut = 0
	failures = 0

	with base.getWritableAdminConn() as conn:
		idTable = api.TableForDef(rd.getById("sentOut"), connection=conn)

		for rec in sorted(recs, key=lambda r: r["updated"]):
			try:
				sendOneToMicroblog(rec, idTable, now, mastodon_service)
				sentOut += 1
			except IOError as msg:	# mishaps when microblogging (try again later)
				sys.stderr.write("Failure for %s: %s\n"%(rec["ivoid"], str(msg)))
				failures += 1
			except Exception: # Internal error
				sys.stderr.write("Oooops, disaster %s\n"%rec["ivoid"])
				failures += 1
				traceback.print_exc()
			if sentOut>=4:
				break
	
	if failures:
		raise base.ReportableError("Some microblog communication failed")


####################### end microblogging interface


import argparse #how to parse command-line arguments

#for debug parse argument
def parseCommandLine():
	parser = argparse.ArgumentParser(description=
		"Make an RSS feed from the relational registry.")
	parser.add_argument("-d", "--debug", dest="debug",
		action="store_true", help="debug mode, don't write"
			" feed but print internal record representation")
	return parser.parse_args()


def main():
	recs = getRecsFromTAP()
	data = makeRSS(recs).render(prefixForEmpty="atom")
	rd = api.getRD("registryrss/q")
	with open(rd.getAbsPath("content/feed.xml.tmp"), "wb") as f:
		f.write(data)
	os.rename(rd.getAbsPath("content/feed.xml.tmp"),
		rd.getAbsPath("content/feed.xml"))
	sendToMicroblog(recs)


if __name__=="__main__":
	main()
