#!/usr/bin/python

import urllib2;
import xml.dom.minidom;
import webbrowser;

################################################################################
#                      http
################################################################################
def queryImdbByTitle(movieTitle):
  """
  make an http call to imdb
  try to find a movie by title
  return non-valid html as string
  """
  url = "http://www.imdb.org/search/text";
  get = "?realm=title&field=plot&q=";
  title = "";
  for word in movieTitle.split(" "):
    title = title + word+"+"
  print url+get+title
  #return urllib2.urlopen(url+get+title).read()

def __queryImdbByPath(path):
  """
  make an http call to imdb
  get movie page from path ex: /title/tt0079501/
  return non-valid html as string
  """
  url = "http://www.imdb.org";
  return urllib2.urlopen(url+path).read()

################################################################################
#                      html to xml
################################################################################
def __extractWellFormedXML(html,begin,end):
  """
  from non valid html extract the part between begin and end
  repace some non xml valid common error with valid one
  return xml as string
  """
  resultstart = html.find(begin);
  resultend = html.find(end,resultstart) + len(end);
  return html[resultstart:resultend].replace("<br>","<br />").replace("&","and");

def __extractWellFormedXMLFromImdbByTitle(html):
  """
  from non valid html extract the part that contain list of movie
  repace some non xml valid common error with valid one
  return xml as string
  """
  begin = "<table class=\"results\">";
  end = "</table>";
  return __extractWellFormedXML(html,begin,end);

def __extractWellFormedXMLFromImdbByUrl(html):
  """
  from non valid html extract the part that contain a movie
  repace some non xml valid common error with valid one
  return xml as string
  """
  begin = "<td id=\"overview-top\">";
  end = "</td>";
  return __extractWellFormedXML(html,begin,end).replace("itemscope","itemscope='itemscope'");

################################################################################
#                      xml dom
################################################################################
def __isElem(elem):
  return elem.nodeType == elem.ELEMENT_NODE

def __iterdom(dom):
  for elem in filter(__isElem , dom.childNodes):
    yield elem;

def __getMoviesFromImdbByTitle(dom):
  """
  from dom object
  yield a list of movie tuple (imdb,title,year,about)
  """
  for movie in dom.getElementsByTagName("tr"):
    title = None; 
    imdb = None;
    year = None;
    about = None;
    for elem in __iterdom(movie):
      if elem.getAttribute("class") == "title":
        title = elem.firstChild.firstChild.nodeValue;
        imdb = elem.firstChild.getAttribute("href");
        for yyear in elem.getElementsByTagName("span"):
          year = yyear.firstChild.nodeValue;
        for aabout in elem.getElementsByTagName("div"):
          about = aabout.firstChild.nodeValue;
    yield (imdb,title,year,about);

def __getMovieFromImdbByUrl(dom):
  """
  from dom object
  return a movie tuple (title,year,about)
  """
  title = None;
  year = None;
  about = "nope";
  for div in __iterdom(dom):
    for elem in __iterdom(div):
      if elem.getAttribute("class") == "header":
        for subelem in __iterdom(elem):
          if subelem.getAttribute("class") == "itemprop":
            title = subelem.firstChild.nodeValue;
          if subelem.getAttribute("class") == "nobr":
            for subsubelem in __iterdom(subelem):
              year = subsubelem.firstChild.nodeValue;
        if elem.tagName == "p":
          if elem.getAttribute("itemprop") == "description":
            about = elem.firstChild.nodeValue;
  return (title,year,about);

################################################################################
#                      front end
################################################################################

def findMovie(title):
  """
  from title
  query imdb
  yield some movie tuple that matche
  """
  _html = __queryImdbByTitle(title);
  _xml = __extractWellFormedXMLFromImdbByTitle(_html);
  _dom = xml.dom.minidom.parseString(_xml);
  return __getMoviesFromImdbByTitle(_dom);

def getMovie(path):
  """
  from path
  query imdb
  return a movie tuple
  """
  _html = __queryImdbByPath(path);
  _xml = __extractWellFormedXMLFromImdbByUrl(_html);
  _dom = xml.dom.minidom.parseString(_xml);
  title,year,about = __getMovieFromImdbByUrl(_dom);
  return (path,title,year,about);

def imdbOpen(path):
  """
  open webbrowser with imdb path
  """
  webbrowser.open("http://www.imdb.org"+path);
################################################################################
#                      test
################################################################################
if __name__ == "__main__":
  import sys;
  
  def findmovieshowlist(movietitle):
    count=0;
    for (imdb,title,year,about) in findMovie(movietitle):
      print "title:",title,year
      print "imdb:",imdb
      print "plot",about
      print
      count += 1;
    print "found",count;

  def showmoviefromurl(imdburl):
    path,title,year,about = getMovie(imdburl);
    print "title:",title,year
    print "imdb:",imdburl
    print "plot",about
    print
  
  def showhelp(nothing=None):
    print "usage:",sys.argv[0],"command, value"
    print "command:"
    print "  help    show this"
    print "  find    find movie on imdb, return list"
    print "  fetch   from imdb path show movie info ex: "
  
  command = {
    "find":findmovieshowlist,
    "fetch":showmoviefromurl,
    "help":showhelp
  };
  
  if len(sys.argv) < 3:
    showhelp();
  elif sys.argv[1] in command:
    command[sys.argv[1]](sys.argv[2] or None);
  else:
    showhelp();
  
  
