#!/usr/bin/env python
"""
A tool to gather info from OAI-PMH registry endpoints serving VOR records
and format them as an RSS stream.
"""

# NOTE: This isn't used any more by GAVO, and you should probably use
# gavo/protocols/oaiclient.py from GAVO DaCHS (http://soft.g-vo.org/dachs)
# rather than this as well.


# This code is in the public domain.

# We want everything in one file for now and thus have "chapters" rather
# than modules.
#
# If you want to use the script on a site different from GAVO DC,
# you must change quite a few items in makeRSS.  Sorry, not much
# sense in making this configurable (yet).
#
# RSS will only work with certain GAVO packages installed.
# Microblogging takes the passwords from GAVO-internal files accessed
# through incredibly complicated ways.  You'll have to change this
# when you want to use it.

import datetime
import os
import random
import re
import sys
import time
import urllib
import warnings
import weakref
from xml import etree
from xml import sax
from xml.sax.handler import ContentHandler


################################ Configuration

REGISTRIES = {
  "local": "http://localhost:8080/oai.xml",
  "gavo": "http://dc.zah.uni-heidelberg.de/oai.xml",
  "vizier": "http://cdsweb.u-strasbg.fr/reg-bin/vizier/oai.pl",
  "nvopub": "http://nvo.ncsa.uiuc.edu/cgi-bin/reg10/oai.pl",
  "astrogrid": "http://rakaposhi.star.le.ac.uk:8080/astrogrid-registry/OAIHandlerv1_0",
  "stscis": "http://nvo.stsci.edu/vor10/oai.aspx",
  "eurovos": "http://registry.euro-vo.org/oai.jsp",
  "caltechs": "http://nvo.caltech.edu:8080/carnivore/cgi-bin/OAI-XML/carnivore/OAI.pl",
}

cacheResults = False

twitterService = "http://twitter.com/statuses/update.xml"
identicaService = "http://identi.ca/api/statuses/update.xml"

################################ Helpers

class FailedQuery(Exception):
  pass


class NoRecordsMatch(Exception):
  pass


class AppURLopener(urllib.FancyURLopener):
  version = "GAVO RSS harvester"
  _credentials = None

  def addCredentials(self, realm, user, password):
    if self._credentials is None:
      self._credentials = {}
    self._credentials[realm] = user, password

  def prompt_user_passwd(self, host, realm):
    if realm in self._credentials:
      return self._credentials[realm]
    raise FailedQuery("%s requires creds for %s, but we don't have any."%
      host, realm)


urlopener = AppURLopener()


def getWithCache(url):
  """retrieves URL and returns it.
  """
  cacheName = re.sub("[^\w]+", "", url)+".cache"
  if cacheResults and os.path.exists(cacheName):
    doc = open(cacheName).read()
  else:
    doc = urlopener.open(url).read()
    if cacheResults:
      f = open(cacheName, "w")
      f.write(doc)
      f.close()
  return doc


class StartEndHandler(ContentHandler):
  """This class provides startElement, endElement and characters
  methods that translate events into method calls.

  When an opening tag is seen, we look of a _start_<element name>
  method and, if present, call it with the name and the attributes.
  When a closing tag is seen, we try to call _end_<element name> with
  name, attributes and contents.  If the _end_xxx method returns a
  string (or similar), this value will be added to the content of the
  enclosing element.
  """
  def __init__(self):
    ContentHandler.__init__(self)
    self.realHandler = weakref.proxy(self)
    self.elementStack = []
    self.contentsStack = [[]]

  def processingInstruction(self, target, data):
    self.contentsStack[-1].append(data)

  def cleanupName(self, name,
      cleanupPat=re.compile(".*:")):  #Nuke namespaces
    return cleanupPat.sub("", name).replace("-", "_")

  def startElement(self, name, attrs):
    self.contentsStack.append([])
    name = self.cleanupName(name)
    self.elementStack.append((name, attrs))
    if hasattr(self.realHandler, "_start_%s"%name):
      getattr(self.realHandler, "_start_%s"%name)(name, attrs)
    elif hasattr(self, "_defaultStart"):
      self._defaultStart(name, attrs)

  def endElement(self, name, suppress=False):
    contents = "".join(self.contentsStack.pop())
    name = self.cleanupName(name)
    _, attrs = self.elementStack.pop()
    res = None
    if hasattr(self.realHandler, "_end_%s"%name):
      res = getattr(self.realHandler,
        "_end_%s"%name)(name, attrs, contents)
    elif hasattr(self, "_defaultEnd"):
      res = self._defaultEnd(name, attrs, contents)
    if isinstance(res, basestring) and not suppress:
      self.contentsStack[-1].append(res)

  def characters(self, chars):
    self.contentsStack[-1].append(chars)

  def getResult(self):
    return self.contentsStack[0][0]

  def getParentTag(self):
    if self.elementStack:
      return self.elementStack[-1][0]


def datetimeToRFC2616(dt):
       """returns a UTC datetime object in the format requried by http.

       This may crap when you fuzz with the locale.
       """
       return dt.strftime('%a, %d %b %Y %H:%M:%S GMT')

################################## Document parsers


def parseISODate(dtSpec):
  """returns a datetime instance from an ISO dtSpec, applying a bit of force.
  """
  # fractional seconds, timezones: only trouble
  dtSpec = re.sub("\.\d\d*Z?$|Z$", "", dtSpec)
  try:
    val = datetime.datetime(*time.strptime(dtSpec, "%Y-%m-%dT%H:%M:%S")[:6])
  except ValueError:
    try:
      val = datetime.datetime(*time.strptime(dtSpec, "%Y-%m-%d")[:3])
    except ValueError:  # hopelessly botched date,
                        # we want a datetime nonetheless
      val = datetime.datetime.utcnow()-datetime.timedelta(days=30)
  return val


class IdParser(StartEndHandler):
  """A parser for simple OAI-PMH headers.

  Records end up as a list of dictionaries in the recs attribute.
  """
  resumptionToken = None

  def __init__(self, initRecs=None):
    StartEndHandler.__init__(self)
    if initRecs is None:
      self.recs = []
    else:
      self.recs = initRecs

  def _end_identifier(self, name, attrs, content):
    self.recs[-1]["id"] = content

  def _end_datestamp(self, name, attrs, content):
    self.recs[-1]["date"] = parseISODate(content)

  def _start_header(self, name, attrs):
    self.recs.append({})

  def _end_error(self, name, attrs, content):
    if attrs["code"]=="noRecordsMatch":
      raise NoRecordsMatch()
    raise FailedQuery("Registry bailed with code %s, value %s"%(
      attrs["code"], content))

  def _end_resumptionToken(self, name, attrs, content):
    if content.strip():
      self.resumptionToken = content


class RecordParser(IdParser):
  """A simple parser for ivo_vor records.
  """
  def _end_title(self, name, attrs, content):
    if self.getParentTag()=="Resource":
      self.recs[-1][name] = content

  def _end_name(self, name, attrs, content):
    if self.getParentTag()=="creator":
      self.recs[-1].setdefault(name, []).append(content)

  def _end_subject(self, name, attrs, content):
    self.recs[-1].setdefault(name, []).append(content)

  def _handleContentChild(self, name, attrs, content):
    if self.getParentTag()=="content":
      self.recs[-1][name] = content

  _end_description = _end_source = _end_referenceURL = \
    _handleContentChild

  def _end_datestamp(self, name, attrs, content):
    # nuke IdParser implementation, we take our date from ri:Resource
    pass

  def _startResource(self, name, attrs):
    self.recs.append({})

  def _end_Resource(self, name, attrs, content):
    self.recs[-1]["date"] = parseISODate(attrs["updated"])

  def _end_accessURL(self, name, attrs, content):
    self.recs[-1].setdefault(name, []).append(content)


################################## OAI interface handling

def getOpQS(**args):
  """returns a properly quoted HTTP query part from its (keyword) arguments.
  """
  # we don't use urllib.urlencode to not encode empty values like a=&b=val
  qString = "&".join("%s=%s"%(k, urllib.quote(v))
    for k, v in args.iteritems() if v)
  return "%s"%(qString)


def getOAIKWs(opts):
  """returns a dictionary containing query keywords for OAI interfaces
  from what's specified on the command line.
  """
  kws = {}
  # XXX TODO: allow for different granularities
  if opts.startDate:
    kws["from"] = opts.startDate.strftime("%Y-%m-%dT%H:%M:%S")
  if opts.endDate:
    kws["until"] = opts.endDate.strftime("%Y-%m-%dT%H:%M:%S")
#  kws["set"] = "ivo_managed"
  return kws


def talkOAI(verb, parserClass, opts):
  """processes an OAI dialogue for verb using the IdParser-derived parserClass.
  """
  baseURL = REGISTRIES[opts.registry]+"?verb=%s&"%verb
  res = getWithCache(baseURL+
    getOpQS(metadataPrefix="ivo_vor", **getOAIKWs(opts)))
  handler = parserClass()
  try:
    sax.parseString(res, handler)
    recs = handler.recs
    while handler.resumptionToken is not None:
      resumptionToken = handler.resumptionToken
      handler = parserClass(recs)
      sax.parseString(getWithCache(baseURL+
        "resumptionToken=%s"%urllib.quote(resumptionToken)),
        handler)
      recs = handler.recs
    return recs
  except NoRecordsMatch:
    return []


def getIdentifiers(opts):
  """returns a list of "short" records for what's in the registry specified
  by opts.
  """
  return talkOAI("ListIdentifiers", IdParser, opts)


def getRecords(opts):
  """returns a list of "long" records for what's in the registry specified
  by opts.
  """
  return talkOAI("ListRecords", RecordParser, opts)


def getNewRecordsFromEuroVO(opts):
  """returns a list of "long" records taken from what EuroVO thinks
  was new with them in the last 30 days.

  opts settings are ignored (we should probably warn if values are
  off-default).
  """
  opts.registry = "eurovos"
  opts.startDate = datetime.datetime.utcnow()-datetime.timedelta(days=20)
  return getRecords(opts)


def getNewRecordsFromAstrogrid(opts):
  """returns a list of "long" records taken from what Astrogrid thinks
  was new with them in the last 30 days.

  opts settings are ignored (we should probably warn if values are
  off-default).
  """
  opts.registry = "astrogrid"
  opts.startDate = datetime.datetime.utcnow()-datetime.timedelta(days=20)
  return getRecords(opts)



############################### RSS building (only works with the GAVO package)

try:

  from gavo.utils import stanxml

  class RSS(object):
    """the namespace for rss elements.
    """
    class RSSElement(stanxml.Element):
      _local = True

    class rss(RSSElement):
      _a_version = "2.0"

    class guid(RSSElement):
      _a_isPermaLink = "false"
    class author(RSSElement): pass
    class category(RSSElement): pass
    class channel(RSSElement): pass
    class description(RSSElement): pass
    class image(RSSElement): pass
    class item(RSSElement): pass
    class language(RSSElement): pass
    class lastBuildDate(RSSElement): pass
    class link(RSSElement): pass
    class pubDate(RSSElement): pass
    class title(RSSElement): pass
    class url(RSSElement): pass

  stanxml.registerPrefix("atom", "http://www.w3.org/2005/Atom", None)

  class Atom(object):
    """the namespace of atom elements.
    """
    class AtomElement(stanxml.Element):
      _prefix = "atom"

    class link(AtomElement):
      _mayBeEmpty = True
      _a_href = None
      _a_rel = None
      _a_type = "application/rss+xml"

except ImportError:  # no gavo.utils, no RSS
  pass


def makeItem(rec):
  if not (rec.get("title") and rec.get("description") and rec.get("accessURL")):
    return

  # HTML is embedded as plain text in the description element.  Thus,
  # we do daring text operations...
  description = ["<dl>",
    "<dt>Description</dt>",
    "<dd>%s</dd>"%rec["description"]]
  if rec.get("name"):
    description.extend([
      "<dt>Author(s)</dt>",
      "<dd>%s</dd>"%", ".join(rec["name"])])
  description.extend([
    "<dt>IVOA id</dt>",
    "<dd>%s</dd>"%rec["id"]])
  description.append("</dl>")

  return RSS.item[
    RSS.title[rec.get("title")],
    RSS.link[rec.get("referenceURL")],
    RSS.guid(isPermaLink="false")[rec["id"]],
    RSS.author["gavo@ari.uni-heidelberg.de (GAVO)"],
    RSS.pubDate[datetimeToRFC2616(rec["date"])],
    RSS.description["".join(description)],
    [RSS.category[kw] for kw in rec.get("subject", [])]]


def makeRSS(recs, opts):
  return RSS.rss[
    RSS.channel[
      Atom.link(href="http://vo.uni-hd.de/regrss", rel="self",
        type="application/rss+xml"),
      RSS.title["VO Fresh"],
      RSS.link["http://vo.uni-hd.de/registryrss/q/rss/info"],
      RSS.description["New services and resources in the Virtual Observatory,"
        "  derived from the Astrogrid registry."],
      RSS.language["en"],
      RSS.lastBuildDate[datetimeToRFC2616(datetime.datetime.utcnow())],
      RSS.image[
        RSS.link["http://vo.uni-hd.de/registryrss/q/rss/info"],
        RSS.title["VO Fresh"],
        # image assumed to be 88x31 pixels
        RSS.url["http://vo.uni-hd.de/registryrss/q/rss/static/logo.png"]],
      [makeItem(rec) for rec in recs]]]


def printRSS(recs, opts):
  sys.stdout.write(makeRSS(recs, opts).render())


############################### Microblogging interface


def truncateToWords(text, maxLength=104, suffix="..."):
  """returns text truncated at whitespace to maxLength char max.
  """
  if len(text)>maxLength:
    pat = re.compile(r"^(.{0,%d}\S)\s.*"%(maxLength-len(suffix)))
    return pat.sub(r"\1", text)+suffix
  else:
    return text


def shortenWithIsGd(url):
  """returns a shortened URL using is.gd.
  """
# this is not used any more since they blocked alnilam after slight
# mishaps on the sides of both us and the EuroVO registry
  shortenService = "http://v.gd/create.php?format=simple&url="
  for i in range(10):
    shortLinkF = urllib.urlopen(shortenService+urllib.quote(url))
    if shortLinkF.getcode()==502: # try again later, see is.gd API
      time.sleep(60)
    else:
      break
  else:
    raise FailedQuery("Shorten service delayed even after %d attempts"%i)
  return shortLinkF.read()


def shorten(url):
  """returns a shortened URL using ur1.ca.
  """
  f = urllib.urlopen("http://ur1.ca/", urllib.urlencode({"longurl": url}))
  try:
    if f.getcode()!=200:
      sys.stderr.write("\n\n"+f.read()+"\n\n")
      raise IOError("shorten URL %s failed (HTTP %s)"%(url, f.getcode()))
    soup = f.read()
  finally:
    f.close()
  # hell, screen scraping (but we don't depend on beautifulsoup!)
  succP = re.search('<p class="success">(.*?)</p>', soup).group(1)
  return re.search('href="([^"]+)"', succP).group(1)


def getMicroblogText(rec):
  """returns a dictionary containing items for representing the
  resource rec in a microblogging service.
  """
  return "New VO Service: %s %s"%(
    shorten(rec["referenceURL"]),
    truncateToWords(rec["title"]))


def sendToMicroblogService(apiURL, text):
  time.sleep(random.randint(5, 10))
  response = urlopener.open(apiURL,
    urllib.urlencode({"status": text})).read()
  if "<error>" in response:
    sys.stderr.write(response)
    warnings.warn("update of '%s' to %s failed"%(text, apiURL))


def sendOneToMicroblog(rec, idTable, now):
  ivo_id = rec["id"]
  try:
    lastPublished = idTable.getRow(ivo_id)["sent"]
  except KeyError:  # new ivo id, not yet in DB.
    lastPublished = None
  except IOError: # shorten service failed
    sys.stderr.write("(shorten) ")
    raise
  if lastPublished:
    return
  try:
    text = getMicroblogText(rec).encode("utf-8", 'ignore')
  except KeyError:  # incomplete record, ignore
    return

  try:
    sendToMicroblogService(twitterService, text)
  except IOError:
    sys.stderr.write("(shorten) ")
    raise

  try:
    sendToMicroblogService(identicaService, text)
  except IOError:
    sys.stderr.write("(shorten) ")
    raise

  idTable.addRow({"ivo_id": ivo_id, "sent": now})


def sendToMicroblogs(recs, opts):
  """sends everything mentioned in recs to twitter and/or identi.ca.

  As a side effect, it updates the sentOut table in the database.

  This requires a working GAVO infrastructure and a matching RD.
  If you need this, talk to the authors.
  """
  from gavo import api
  rd = api.getRD("registryrss/q")
  urlopener.addCredentials("Twitter API",
    rd.getProperty("twitterUID"), rd.getProperty("twitterPW"))
  urlopener.addCredentials("Identi.ca API",
    rd.getProperty("identiUID"), rd.getProperty("identiPW"))
  now = datetime.datetime.utcnow()
  idTable = api.TableForDef(rd.getById("sentOut"))
  try:
    for rec in sorted(recs, key=lambda r: r["date"]):
      try:
        sendOneToMicroblog(rec, idTable, now)
      except IOError, msg:  # mishaps when microblogging (try again later)
        sys.stderr.write("Failure for %s: %s\n"%(rec["id"], str(msg)))
  finally:
    idTable.commit().close()


############################### User interface, I/O


def listIdentifiers(recs, opts):
  """prints short format records.

  data suitable from recs is returned by both getIdentifers and getRecords.
  """
  for rec in recs:
    if "date" in rec:
      print rec["date"], rec["id"]
    else:
      print "UNDATED     ", rec["id"]


def prettyPrint(recs, opts):
  import pprint
  pprint.pprint(recs)


def _enableCaching(*args):
  global cacheResults
  cacheResults = True


def _getExtendedOptionClass():
  from optparse import Option, OptionValueError
  from copy import copy

  def checkDatetime(option, opt, val):
    try:
      return datetime.datetime(*time.strptime(val, "%Y-%m-%dT%H:%M:%S")[:6])
    except ValueError:
      try:
        return datetime.datetime(*time.strptime(val, "%Y-%m-%d")[:3])
      except ValueError:
        raise OptionValueError(
          "option %s: Dates/Datetimes have to be in"
            " ISO format without time zone"%option)

  class OptionWithDate(Option):
    TYPES = Option.TYPES+("date",)
    TYPE_CHECKER = copy(Option.TYPE_CHECKER)
    TYPE_CHECKER["date"] = checkDatetime

  return OptionWithDate


def defaultAction(recs, opts):
  printRSS(recs, opts)
  #sendToMicroblogs(recs, opts)


ACTIONS = {
  "ls": (getIdentifiers, listIdentifiers),
  "dump": (getRecords, prettyPrint),
  "rss": (getRecords, printRSS),
  "microblog": (getRecords, sendToMicroblogs),
  "default": (getNewRecordsFromEuroVO, defaultAction),
}


def parseCommandLine():
  from optparse import OptionParser
  parser = OptionParser(usage="%%prog [options] <action>\n where action"
    " is one of %s"%(", ".join(ACTIONS)),
    option_class=_getExtendedOptionClass())
  parser.add_option("-n", "--newer-than", help="Only request records newer"
    " than ISODATE.", metavar="ISODATE", action="store", default=None,
    type="date", dest="startDate")
  parser.add_option("-o", "--older-than", help="Only request records older"
    " than ISODATE.", metavar="ISODATE", action="store", default=None,
    type="date", dest="endDate")
  parser.add_option("-r", "--registry", help="Use registry of PROVIDER"
    " (help to see selection)", metavar="PROVIDER", action="store",
    default="gavo", dest="registry")
  parser.add_option("-c", "--cache", help="Turn on response caching"
    " (for debugging only)", action="callback", callback=_enableCaching)
  opts, args = parser.parse_args()
  if opts.registry=="help" or opts.registry not in REGISTRIES:
    sys.exit("Invalid registry %s.  Available keywords: %s"%(
      opts.registry, ", ".join(REGISTRIES)))
  if len(args)!=1 or args[0] not in ACTIONS:
    parser.print_help()
    sys.exit(1)
  action = args[0]
  return opts, action


def main():
  opts, action = parseCommandLine()
  getData, formatData = ACTIONS[action]
  data = [r for r in getData(opts) if "date" in r]
  data.sort(key=lambda r: r["date"])
  data.reverse()
  # avoid huge RSS documents
  data = data[:30]
  formatData(data, opts)


if __name__=="__main__":
  main()

# vi:et:ts=2:sta:
