summaryrefslogtreecommitdiff
path: root/freebase
diff options
context:
space:
mode:
authorkientzle <kientzle@5914aa95-5b3a-0410-a3b5-7b719e7fe9b2>2008-03-18 18:24:32 +0000
committerkientzle <kientzle@5914aa95-5b3a-0410-a3b5-7b719e7fe9b2>2008-03-18 18:24:32 +0000
commita9aa9e93ab46f067bfe912aae3612a17f2743720 (patch)
treebed621f2d5dda9eace6fede1b083a3c09ad8e980 /freebase
parent1abc18880a041188e8259bc637c437a4d3780526 (diff)
Move the "freebase.api" module into a "freebase-api" directory so that
we can provide a couple of different parallel implementations. git-svn-id: http://freebase-python.googlecode.com/svn/trunk@34 5914aa95-5b3a-0410-a3b5-7b719e7fe9b2
Diffstat (limited to 'freebase')
-rwxr-xr-xfreebase/__init__.py0
-rwxr-xr-xfreebase/api/__init__.py4
-rwxr-xr-xfreebase/api/httplib2cookie.py149
-rwxr-xr-xfreebase/api/mqlkey.py129
-rw-r--r--freebase/api/session.py524
-rwxr-xr-xfreebase/rison.py308
-rwxr-xr-xfreebase/uritemplate.py143
7 files changed, 0 insertions, 1257 deletions
diff --git a/freebase/__init__.py b/freebase/__init__.py
deleted file mode 100755
index e69de29..0000000
--- a/freebase/__init__.py
+++ /dev/null
diff --git a/freebase/api/__init__.py b/freebase/api/__init__.py
deleted file mode 100755
index c3adebc..0000000
--- a/freebase/api/__init__.py
+++ /dev/null
@@ -1,4 +0,0 @@
-
-from session import HTTPMetawebSession, MetawebError, attrdict
-
-from mqlkey import quotekey, unquotekey
diff --git a/freebase/api/httplib2cookie.py b/freebase/api/httplib2cookie.py
deleted file mode 100755
index fee0ec0..0000000
--- a/freebase/api/httplib2cookie.py
+++ /dev/null
@@ -1,149 +0,0 @@
-# ========================================================================
-# Copyright (c) 2007, Metaweb Technologies, Inc.
-# All rights reserved.
-#
-# Redistribution and use in source and binary forms, with or without
-# modification, are permitted provided that the following conditions
-# are met:
-# * Redistributions of source code must retain the above copyright
-# notice, this list of conditions and the following disclaimer.
-# * Redistributions in binary form must reproduce the above
-# copyright notice, this list of conditions and the following
-# disclaimer in the documentation and/or other materials provided
-# with the distribution.
-#
-# THIS SOFTWARE IS PROVIDED BY METAWEB TECHNOLOGIES AND CONTRIBUTORS
-# ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
-# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
-# FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL METAWEB
-# TECHNOLOGIES OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
-# INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
-# BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
-# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
-# CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
-# LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
-# ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
-# POSSIBILITY OF SUCH DAMAGE.
-# ========================================================================
-
-#
-#
-# httplib2cookie.py allows you to use python's standard
-# CookieJar class with httplib2.
-#
-#
-
-import re
-import httplib2
-from httplib2 import Http
-
-import mimetools
-import urllib
-import urllib2
-import cookielib
-
-class DummyRequest(object):
- """Simulated urllib2.Request object for httplib2
-
- implements only what's necessary for cookielib.CookieJar to work
- """
- def __init__(self, url, headers=None):
- self.url = url
- self.headers = headers
- self.origin_req_host = cookielib.request_host(self)
- self.type, r = urllib.splittype(url)
- self.host, r = urllib.splithost(r)
- if self.host:
- self.host = urllib.unquote(self.host)
-
- def get_full_url(self):
- return self.url
-
- def get_origin_req_host(self):
- # TODO to match urllib2 this should be different for redirects
- return self.origin_req_host
-
- def get_type(self):
- return self.type
-
- def get_host(self):
- return self.host
-
- def get_header(self, key, default=None):
- return self.headers.get(key.lower(), default)
-
- def has_header(self, key):
- return key in self.headers
-
- def add_unredirected_header(self, key, val):
- # TODO this header should not be sent on redirect
- self.headers[key.lower()] = val
-
- def is_unverifiable(self):
- # TODO to match urllib2, this should be set to True when the
- # request is the result of a redirect
- return False
-
-
-class DummyResponse(object):
- """Simulated urllib2.Request object for httplib2
-
- implements only what's necessary for cookielib.CookieJar to work
- """
- def __init__(self, response):
- self.response = response
-
- def info(self):
- return DummyMessage(self.response)
-
-
-class DummyMessage(object):
- """Simulated mimetools.Message object for httplib2
-
- implements only what's necessary for cookielib.CookieJar to work
- """
- def __init__(self, response):
- self.response = response
-
- def getheaders(self, k):
- k = k.lower()
- v = self.response.get(k.lower(), None)
- if k not in self.response:
- return []
- #return self.response[k].split(re.compile(',\\s*'))
-
- # httplib2 joins multiple values for the same header
- # using ','. but the netscape cookie format uses ','
- # as part of the expires= date format. so we have
- # to split carefully here - header.split(',') won't do it.
- HEADERVAL= re.compile(r'\s*(([^,]|(,\s*\d))+)')
- return [h[0] for h in HEADERVAL.findall(self.response[k])]
-
-class CookiefulHttp(Http):
- """Subclass of httplib2.Http that keeps cookie state
-
- constructor takes an optional cookiejar=cookielib.CookieJar
-
- currently this does not handle redirects completely correctly:
- if the server redirects to a different host the original
- cookies will still be sent to that host.
- """
- def __init__(self, cookiejar=None, **kws):
- # note that httplib2.Http is not a new-style-class
- Http.__init__(self, **kws)
- if cookiejar is None:
- cookiejar = cookielib.CookieJar()
- self.cookiejar = cookiejar
-
- def request(self, uri, **kws):
- headers = kws.pop('headers', None)
- req = DummyRequest(uri, headers)
- self.cookiejar.add_cookie_header(req)
- headers = req.headers
-
- (r, body) = Http.request(self, uri, headers=headers, **kws)
-
- resp = DummyResponse(r)
- self.cookiejar.extract_cookies(resp, req)
-
- return (r, body)
diff --git a/freebase/api/mqlkey.py b/freebase/api/mqlkey.py
deleted file mode 100755
index 98b7d83..0000000
--- a/freebase/api/mqlkey.py
+++ /dev/null
@@ -1,129 +0,0 @@
-# ========================================================================
-# Copyright (c) 2007, Metaweb Technologies, Inc.
-# All rights reserved.
-#
-# Redistribution and use in source and binary forms, with or without
-# modification, are permitted provided that the following conditions
-# are met:
-# * Redistributions of source code must retain the above copyright
-# notice, this list of conditions and the following disclaimer.
-# * Redistributions in binary form must reproduce the above
-# copyright notice, this list of conditions and the following
-# disclaimer in the documentation and/or other materials provided
-# with the distribution.
-#
-# THIS SOFTWARE IS PROVIDED BY METAWEB TECHNOLOGIES AND CONTRIBUTORS
-# ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
-# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
-# FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL METAWEB
-# TECHNOLOGIES OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
-# INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
-# BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
-# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
-# CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
-# LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
-# ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
-# POSSIBILITY OF SUCH DAMAGE.
-# ========================================================================
-
-import string
-import re
-
-def quotekey(ustr):
- """
- quote a unicode string to turn it into a valid namespace key
-
- """
- valid_always = string.ascii_letters + string.digits
- valid_interior_only = valid_always + '_-'
-
- if isinstance(ustr, str):
- s = unicode(ustr,'utf-8')
- elif isinstance(ustr, unicode):
- s = ustr
- else:
- raise ValueError, 'quotekey() expects utf-8 string or unicode'
-
- output = []
- if s[0] in valid_always:
- output.append(s[0])
- else:
- output.append('$%04X' % ord(s[0]))
-
- for c in s[1:-1]:
- if c in valid_interior_only:
- output.append(c)
- else:
- output.append('$%04X' % ord(c))
-
- if len(s) > 1:
- if s[-1] in valid_always:
- output.append(s[-1])
- else:
- output.append('$%04X' % ord(s[-1]))
-
- return str(''.join(output))
-
-
-def unquotekey(key, encoding=None):
- """
- unquote a namespace key and turn it into a unicode string
- """
-
- valid_always = string.ascii_letters + string.digits
-
- output = []
- i = 0
- while i < len(key):
- if key[i] in valid_always:
- output.append(key[i])
- i += 1
- elif key[i] in '_-' and i != 0 and i != len(key):
- output.append(key[i])
- i += 1
- elif key[i] == '$' and i+4 < len(key):
- # may raise ValueError if there are invalid characters
- output.append(unichr(int(key[i+1:i+5],16)))
- i += 5
- else:
- raise ValueError, "unquote key saw invalid character '%s' at position %d" % (key[i], i)
-
- ustr = u''.join(output)
-
- if encoding is None:
- return ustr
-
- return ustr.encode(encoding)
-
-
-
-def id_to_urlid(id):
- """
- convert a mql id to an id suitable for embedding in a url path.
- """
-
- # XXX shouldn't be in metaweb.api!
- from mw.formats.http import urlencode_pathseg
-
- segs = id.split('/')
-
- assert isinstance(id, str) and id != '', 'bad id "%s"' % id
-
- if id[0] == '~':
- assert len(segs) == 1
- # assume valid, should check
- return id
-
- if id[0] == '#':
- assert len(segs) == 1
- # assume valid, should check
- return '%23' + id[1:]
-
- if id[0] != '/':
- raise ValueError, 'unknown id format %s' % id
-
- # ok, we have a slash-path
- # requote components as keys and rejoin.
- # urlids do not have leading slashes!!!
- return '/'.join(urlencode_pathseg(unquotekey(seg)) for seg in segs[1:])
-
diff --git a/freebase/api/session.py b/freebase/api/session.py
deleted file mode 100644
index fc32676..0000000
--- a/freebase/api/session.py
+++ /dev/null
@@ -1,524 +0,0 @@
-# ========================================================================
-# Copyright (c) 2007, Metaweb Technologies, Inc.
-# All rights reserved.
-#
-# Redistribution and use in source and binary forms, with or without
-# modification, are permitted provided that the following conditions
-# are met:
-# * Redistributions of source code must retain the above copyright
-# notice, this list of conditions and the following disclaimer.
-# * Redistributions in binary form must reproduce the above
-# copyright notice, this list of conditions and the following
-# disclaimer in the documentation and/or other materials provided
-# with the distribution.
-#
-# THIS SOFTWARE IS PROVIDED BY METAWEB TECHNOLOGIES AND CONTRIBUTORS
-# ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
-# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
-# FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL METAWEB
-# TECHNOLOGIES OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
-# INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
-# BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
-# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
-# CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
-# LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
-# ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
-# POSSIBILITY OF SUCH DAMAGE.
-# ========================================================================
-
-#
-# declarations for external metaweb api.
-#
-#
-# from metaweb.api import HTTPMetawebSession
-#
-# mss = HTTPMetawebSession('sandbox.freebase.com')
-# print mss.mqlread([dict(name=None, type='/type/type')])
-#
-#
-#
-
-__all__ = ['MetawebError', 'MetawebSession', 'HTTPMetawebSession', 'attrdict']
-__version__ = '0.1'
-
-import os, sys, re
-import urllib2
-import cookielib
-import simplejson
-from urllib import quote as urlquote
-import pprint
-import socket
-import logging
-
-try:
- import httplib2
- from httplib2cookie import CookiefulHttp
-except ImportError:
- httplib2 = None
- CookiefulHttp = None
- print ('freebase.api: you can install httplib2 for better performance')
-
-import simplejson.encoder
-# remove whitespace from json encoded output
-simplejson.JSONEncoder.item_separator = ','
-simplejson.JSONEncoder.key_separator = ':'
-# don't escape slashes, we're not pasting into script tags here.
-if simplejson.encoder.ESCAPE_DCT.get('/', None) == r'\/':
- simplejson.encoder.ESCAPE_DCT['/'] = '/'
-
-def urlencode_weak(s):
- return urlquote(s, safe=',/:$')
-
-
-# from http://aspn.activestate.com/ASPN/Cookbook/Python/Recipe/361668
-class attrdict(dict):
- """A dict whose items can also be accessed as member variables.
-
- >>> d = attrdict(a=1, b=2)
- >>> d['c'] = 3
- >>> print d.a, d.b, d.c
- 1 2 3
- >>> d.b = 10
- >>> print d['b']
- 10
-
- # but be careful, it's easy to hide methods
- >>> print d.get('c')
- 3
- >>> d['get'] = 4
- >>> print d.get('a')
- Traceback (most recent call last):
- TypeError: 'int' object is not callable
- """
- def __init__(self, *args, **kwargs):
- dict.__init__(self, *args, **kwargs)
- self.__dict__ = self
-
-
-
-# TODO expose the common parts of the result envelope
-class MetawebError(Exception):
- """
- an error report from the metaweb service.
- """
- pass
-
-
-
-# TODO right now this is a completely unnecessary superclass.
-# is there enough common behavior between session types
-# to justify it?
-class MetawebSession(object):
- """
- MetawebSession is the base class for MetawebSession, subclassed for
- different connection types. Only http is available externally.
-
- This is more of an interface than a class
- """
-
- # interface definition here...
-
-
-# from httplib2
-NORMALIZE_SPACE = re.compile(r'(?:\r\n)?[ \t]+')
-def _normalize_headers(headers):
- return dict([ (key.lower(), NORMALIZE_SPACE.sub(value, ' ').strip()) for (key, value) in headers.iteritems()])
-
-
-class HTTPMetawebSession(MetawebSession):
- """
- a MetawebSession is a request/response queue.
-
- this version uses the HTTP api, and is synchronous.
- """
- # share cookies across sessions, so that different sessions can
- # see each other's writes immediately.
- _default_cookiejar = cookielib.CookieJar()
-
- def __init__(self, service_url, username=None, password=None, prev_session=None, cookiejar=None, cookiefile=None):
- """
- create a new MetawebSession for interacting with the Metaweb.
-
- a new session will inherit state from prev_session if present,
- """
- super(HTTPMetawebSession, self).__init__()
-
- self.log = logging.getLogger()
-
- assert not service_url.endswith('/')
- if not '/' in service_url: # plain host:port
- service_url = 'http://' + service_url
-
- self.service_url = service_url
-
- self.username = username
- self.password = password
-
- self.tid = None
-
- if prev_session:
- self.service_url = prev.service_url
-
- if cookiefile is not None:
- cookiejar = self.open_cookie_file(cookiefile)
-
- if cookiejar is not None:
- self.cookiejar = cookiejar
- elif prev_session:
- self.cookiejar = prev_session.cookiejar
- else:
- self.cookiejar = self._default_cookiejar
-
- if CookiefulHttp is not None:
- self.httpclient = CookiefulHttp(cookiejar=self.cookiejar)
- else:
- cookiespy = urllib2.HTTPCookieProcessor(self.cookiejar)
- self.opener = urllib2.build_opener(cookiespy)
-
-
- def open_cookie_file(self, cookiefile=None):
- if cookiefile is None or cookiefile == '':
- if os.environ.has_key('HOME'):
- cookiefile = os.path.join(os.environ['HOME'], '.pyfreebase/cookiejar')
- else:
- raise MetawebError("no cookiefile specified and no $HOME/.pyfreebase directory" % cookiefile)
-
- cookiejar = cookielib.LWPCookieJar(cookiefile)
- if os.path.exists(cookiefile):
- cookiejar.load(ignore_discard=True)
-
- return cookiejar
-
-
- def _httpreq(self, service_path, method='GET', body=None, form=None,
- headers=None):
- """
- make an http request to the service.
-
- form arguments are encoded in the url, even for POST, if a non-form
- content-type is given for the body.
-
- returns a pair (resp, body)
-
- resp is the response object and may be different depending
- on whether urllib2 or httplib2 is in use?
- """
-
- if method == 'POST':
- assert body is not None or form is not None
- elif method == 'GET':
- assert body is None
- else:
- assert 0, 'unknown method %s' % method
-
- url = self.service_url + service_path
-
- if headers is None:
- headers = {}
- else:
- headers = _normalize_headers(headers)
-
- # XXX This is a lousy way to parse Content-Type, where is
- # the library?
- ct = headers.get('content-type', None)
- if ct is not None:
- ct = ct.split(';')[0]
-
- if body is not None:
- # if body is provided, content-type had better be too
- assert ct is not None
-
- if form is not None:
- qstr = '&'.join(['%s=%s' % (urlencode_weak(k), urlencode_weak(v))
- for k,v in form.items()])
- if method == 'POST':
- # put the args on the url if we're putting something else
- # in the body. this is used to add args to raw uploads.
- if body is not None:
- url += '?' + qstr
- else:
- if ct is None:
- # XXX encoding and stuff
- ct = 'application/x-www-form-urlencoded'
- headers['content-type'] = ct
-
- if ct == 'multipart/form-encoded':
- # XXX fixme
- raise NotImplementedError
- elif ct == 'application/x-www-form-urlencoded':
- body = qstr
- else:
- # for all methods other than POST, use the url
- url += '?' + qstr
-
-
- # assure the service that this isn't a CSRF form submission
- headers['x-metaweb-request'] = 'Python'
-
- if 'user-agent' not in headers:
- headers['user-agent'] = 'python freebase.api-%s' % __version__
-
- #if self.tid is not None:
- # headers['x-metaweb-tid'] = self.tid
-
- ####### DEBUG MESSAGE - should check log level before generating
- if form is None:
- formstr = ''
- else:
- formstr = 'FORM:\n ' + '\n '.join(['%s=%s' % (k,v)
- for k,v in form.items()])
- if headers is None:
- headerstr = ''
- else:
- headerstr = 'HEADERS:\n ' + '\n '.join([('%s: %s' % (k,v))
- for k,v in headers.items()])
- self.log.debug('%s %s%s%s', method, url, formstr, headerstr)
- #######
-
- if CookiefulHttp is not None:
- return self._httplib2_request(url, method, body, headers)
- else:
- return self._urllib2_request(url, method, body, headers)
-
-
- def _raise_service_error(self, status, ctype, body):
- is_jsbody = (e.info().type.endswith('javascript')
- or e.info().type.endswith('json'))
- if str(status) == '400' and is_jsbody:
- r = self._loadjson(body)
- msg = r.messages[0]
- raise MetawebError(u'%s %s %r' % (msg.get('code',''), msg.message, msg.info))
-
- raise MetawebError, 'request failed: %s: %r %r' % (url, str(e), body)
-
- def _urllib2_request(self, url, method, body, headers):
- req = urllib2.Request(url, body, headers)
-
- try:
- resp = self.opener.open(req)
-
- except socket.error, e:
- self.log.error('SOCKET FAILURE: %s', e.fp.read())
- raise MetawebError, 'failed contacting %s: %s' % (url, str(e))
-
- except urllib2.HTTPError, e:
- _raise_service_error(e.code, e.info().type, e.fp.read())
-
- for header in resp.info().headers:
- self.log.debug('HTTP HEADER %s', header)
- name, value = re.split("[:\n\r]", header, 1)
- if name.lower() == 'x-metaweb-tid':
- self.tid = value.strip()
-
- return (resp, resp.read())
-
- def _httplib2_request(self, url, method, body, headers):
- try:
- resp, content = self.httpclient.request(url, method=method,
- body=body, headers=headers)
- except socket.error, e:
- self.log.error('SOCKET FAILURE: %s', e.fp.read())
- raise MetawebError, 'failed contacting %s: %s' % (url, str(e))
-
- except httplib2.HttpLib2ErrorWithResponse, e:
- self._raise_service_error(resp.status, resp['content-type'], content)
- except httplib2.HttpLib2Error, e:
- raise MetawebError(u'HTTP error: %s' % (e,))
-
- #tid = resp.get('x-metaweb-tid', None)
-
- return (resp, content)
-
-
- def _httpreq_json(self, *args, **kws):
- resp, body = self._httpreq(*args, **kws)
- return self._loadjson(body)
-
- def _loadjson(self, json):
- # TODO really this should be accomplished by hooking
- # simplejson to create attrdicts instead of dicts.
- def struct2attrdict(st):
- """
- copy a json structure, turning all dicts into attrdicts.
-
- copying descends instances of dict and list, including subclasses.
- """
- if isinstance(st, dict):
- return attrdict([(k,struct2attrdict(v)) for k,v in st.items()])
- if isinstance(st, list):
- return [struct2attrdict(li) for li in st]
- return st
-
- if json == '':
- self.log.error('the empty string is not valid json')
- raise MetawebError('the empty string is not valid json')
-
- try:
- r = simplejson.loads(json)
- except ValueError, e:
- self.log.error('error parsing json string %r' % json)
- raise MetawebError, 'error parsing JSON string: %s' % e
-
- return struct2attrdict(r)
-
- def _check_mqlerror(self, r):
- if r.code != '/api/status/ok':
- for msg in r.messages:
- self.log.error('mql error: %s %s %r' % (msg.code, msg.message, msg.get('query', None)))
- raise MetawebError, 'query failed: %s %r' % (r.messages[0].code, r.messages[0].get('query', None))
-
- def _mqlresult(self, r):
- self._check_mqlerror(r)
-
- # should check log level to avoid redundant simplejson.dumps
- rstr = simplejson.dumps(r.result, indent=2)
- if rstr[0] == '{':
- rstr = rstr[1:-2]
- self.log.info('result: %s', rstr)
-
- return r.result
-
-
-
- def login(self):
- """sign in to the service"""
-
- assert self.username is not None
- assert self.password is not None
-
- self.log.debug('LOGIN USERNAME: %s', self.username)
-
- try:
- r = self._httpreq_json('/api/account/login', 'POST',
- form=dict(username=self.username,
- password=self.password))
- except urllib2.HTTPError, e:
- raise MetawebError("login error: %s", e)
-
- if r.code != '/api/status/ok':
- raise MetawebError(u'%s %r' % (r.get('code',''), r.messages))
-
- self.log.debug('LOGIN RESP: %r', r)
- self.log.debug('LOGIN COOKIES: %s', self.cookiejar)
-
-
- def mqlreaditer(self, sq):
- """read a structure query"""
-
- cursor = True
-
- while 1:
- subq = dict(query=[sq], cursor=cursor, escape=False)
- qstr = simplejson.dumps(subq)
-
- service = '/api/service/mqlread'
-
- r = self._httpreq_json(service, form=dict(query=qstr))
-
- for item in self._mqlresult(r):
- yield item
-
- if r['cursor']:
- cursor = r['cursor']
- self.log.info('CONTINUING with %s', cursor)
- else:
- return
-
- def mqlread(self, sq):
- """read a structure query"""
- subq = dict(query=sq, escape=False)
- if isinstance(sq, list):
- subq['cursor'] = True
-
- service = '/api/service/mqlread'
-
- # should check log level to avoid redundant simplejson.dumps
- self.log.info('%s: %s',
- service,
- simplejson.dumps(sq, indent=2)[1:-2])
-
- qstr = simplejson.dumps(subq)
- r = self._httpreq_json(service, form=dict(query=qstr))
-
- return self._mqlresult(r)
-
- def trans(self, guid):
- """translate blob from guid """
- url = '/api/trans/raw' + urlquote(guid)
-
- self.log.info(url)
-
- resp, body = self._httpreq(url)
-
- self.log.info('%d bytes' % len(body))
-
- return body
-
- def mqlwrite(self, sq):
- """do a mql write"""
- query = dict(query=sq, escape=False)
- qstr = simplejson.dumps(query)
-
- self.log.debug('MQLWRITE: %s', qstr)
-
- service = '/api/service/mqlwrite'
-
- # should check log level to avoid redundant simplejson.dumps
- self.log.info('%s: %s',
- service,
- simplejson.dumps(sq, indent=2)[1:-2])
-
- r = self._httpreq_json(service, 'POST',
- form=dict(query=qstr))
-
- self.log.debug('MQLWRITE RESP: %r', r)
- return self._mqlresult(r)
-
- def mqlflush(self):
- """ask the service not to hand us old data"""
- self.log.debug('MQLFLUSH')
-
- service = '/api/service/mqlwrite'
- r = self._httpreq_json(service, 'POST', form={})
-
- self._check_mqlerror(r)
- return r
-
- def upload(self, body, content_type, document_id=False):
- """upload to the metaweb"""
-
- service = '/api/service/upload'
-
- self.log.info('POST %s: %s (%d bytes)',
- service, content_type, len(body))
-
-
- headers = {}
- if content_type is not None:
- headers['content-type'] = content_type
-
- form = None
- if document_id is not False:
- if document_id is None:
- form = { 'document': '' }
- else:
- form = { 'document': document_id }
-
- # note the use of both body and form.
- # form parameters get encoded into the URL in this case
- r = self._httpreq_json(service, 'POST',
- headers=headers, body=body, form=form)
- return self._mqlresult(r)
-
-
-if __name__ == '__main__':
- console = logging.StreamHandler()
- console.setLevel(logging.DEBUG)
-
- mss = HTTPMetawebSession('sandbox.freebase.com')
-
- self.mss.log.setLevel(logging.DEBUG)
- self.mss.log.addHandler(console)
-
-
- print mss.mqlread([dict(name=None, type='/type/type')])
diff --git a/freebase/rison.py b/freebase/rison.py
deleted file mode 100755
index 2f4eebc..0000000
--- a/freebase/rison.py
+++ /dev/null
@@ -1,308 +0,0 @@
-#
-# rison for python (parser only so far)
-# see http://mjtemplate.org/examples/rison.html for more info
-#
-
-######################################################################
-#
-# the rison parser is based on javascript openlaszlo-json:
-# Author: Oliver Steele
-# Copyright: Copyright 2006 Oliver Steele. All rights reserved.
-# Homepage: http:#osteele.com/sources/openlaszlo/json
-# License: MIT License.
-# Version: 1.0
-#
-
-# hacked by nix for use in uris
-# ported to python by nix
-#
-# TODO
-#
-# switch to unicode
-# fall through to simplejson if first char is not in '!(' -
-# this allows code to use just one parser
-#
-
-
-import os, sys, re
-#import simplejson
-simplejson = None
-
-class ParserException(Exception):
- pass
-
-class Parser(object):
- WHITESPACE = ''
- #WHITESPACE = " \t\n\r\f"
-
- # we divide the uri-safe glyphs into three sets
- # <rison> and <reserved> classes are illegal in ids.
- # <rison> - used by rison (possibly later)
- # <reserved> - not common in strings, reserved
- #not_idchar = "'!=:(),*@$;&";
-
- idchar_punctuation = '_-./~'
- not_idchar = ''.join([c for c in (chr(i) for i in range(127))
- if not (c.isalnum()
- or c in idchar_punctuation)])
-
- # additionally, we need to distinguish ids and numbers by first char
- not_idstart = "-0123456789";
-
- # regexp string matching a valid id
- idrx = ('[^' + not_idstart + not_idchar +
- '][^' + not_idchar + ']*')
-
- # regexp to check for valid rison ids
- id_ok_re = re.compile('^' + idrx + '$', re.M)
-
- # regexp to find the end of an id when parsing
- next_id_re = re.compile(idrx, re.M)
-
- def parse_json(self, str):
- if len(str) > 0 and str[0] not in '!(':
- return simplejson.loads(str)
- return self.parse(str)
-
- def parse(self, str):
- self.string = str
- self.index = 0
-
- value = self.readValue()
- if self.next():
- raise ParserException("unable to parse rison string %r" % (str,))
- return value
-
- def readValue(self):
- c = self.next()
-
- if c == '!':
- return self.parse_bang()
- if c == '(':
- return self.parse_open_paren()
- if c == "'":
- return self.parse_single_quote()
- if c in '-0123456789':
- return self.parse_number()
-
- # fell through table, parse as an id
- s = self.string
- i = self.index-1
-
- m = self.next_id_re.match(s, i)
- if m:
- id = m.group(0)
- self.index = i + len(id)
- return id # a string
-
- if c:
- raise ParserException("invalid character: '" + c + "'")
- raise ParserException("empty expression")
-
- def parse_array(self):
- ar = []
- while 1:
- c = self.next()
- if c == ')':
- return ar
-
- if c is None:
- raise ParserException("unmatched '!('")
-
- if len(ar):
- if c != ',':
- raise ParserException("missing ','")
- elif c == ',':
- raise ParserException("extra ','")
- else:
- self.index -= 1
- n = self.readValue()
- ar.append(n)
-
- return ar
-
- def parse_bang (self):
- s = self.string
- c = s[self.index]
- self.index += 1
- if c is None:
- raise ParserException('"!" at end of input')
- if c not in self.bangs:
- raise ParserException('unknown literal: "!' + c + '"')
- x = self.bangs[c]
- if callable(x):
- return x(self)
-
- return x
-
-
- def parse_open_paren (self):
- count = 0
- o = {}
-
- while 1:
- c = self.next()
- if c == ')':
- return o
- if count:
- if c != ',':
- raise ParserException("missing ','")
- elif c == ',':
- raise ParserException("extra ','")
- else:
- self.index -= 1
- k = self.readValue()
-
- if self.next() != ':':
- raise ParserException("missing ':'")
- v = self.readValue()
-
- o[k] = v
- count += 1
-
-
- def parse_single_quote(self):
- s = self.string
- i = self.index
- start = i
- segments = []
-
- while 1:
- if i >= len(s):
- raise ParserException('unmatched "\'"')
-
- c = s[i]
- i += 1
- if c == "'":
- break
-
- if c == '!':
- if start < i-1:
- segments.append(s[start:i-1])
- c = s[i]
- i += 1
- if c in "!'":
- segments.append(c)
- else:
- raise ParserException('invalid string escape: "!'+c+'"')
-
- start = i
-
-
- if start < i-1:
- segments.append(s[start:i-1])
- self.index = i
- return ''.join(segments)
-
-
- # Also any number start (digit or '-')
- def parse_number(self):
- s = self.string
- i = self.index
- start = i-1
- state = 'int'
- permittedSigns = '-'
- transitions = {
- 'int+.': 'frac',
- 'int+e': 'exp',
- 'frac+e': 'exp'
- }
-
- while 1:
- if i >= len(s):
- i += 1
- break
-
- c = s[i]
- i += 1
-
- if '0' <= c and c <= '9':
- continue
-
- if permittedSigns.find(c) >= 0:
- permittedSigns = ''
- continue
-
- state = transitions.get(state + '+' + c.lower(), None)
- if state is None:
- break
- if state == 'exp':
- permittedSigns = '-'
-
- self.index = i - 1
- s = s[start:self.index]
- if s == '-':
- raise ParserException("invalid number")
- if re.search('[.e]', s):
- return float(s)
- return int(s)
-
- # return the next non-whitespace character, or undefined
- def next(self):
- l = len(self.string)
- s = self.string
- i = self.index
-
- while 1:
- if i == len(s):
- return None
- c = s[i]
- i += 1
- if c not in self.WHITESPACE:
- break
-
- self.index = i
- return c
-
-
- bangs = {
- 't': True,
- 'f': False,
- 'n': None,
- '(': parse_array
- }
-
-
-def loads(s):
- return Parser().parse(s)
-
-if __name__ == '__main__':
- p = Parser()
-
- rison_examples = [
- "(a:0,b:1)",
- "(a:0,b:foo,c:'23skidoo')",
- "!t",
- "!f",
- "!n",
- "''",
- "0",
- "1.5",
- "-3",
- "1e30",
- "1e-30",
- "G.",
- "a",
- "'0a'",
- "'abc def'",
- "()",
- "(a:0)",
- "(id:!n,type:/common/document)",
- "!()",
- "!(!t,!f,!n,'')",
- "'-h'",
- "a-z",
- "'wow!!'",
- "domain.com",
- "'user@domain.com'",
- "'US $10'",
- "'can!'t'",
- ];
-
- for s in rison_examples:
- print
- print '*'*70
- print
- print s
-
- print '%r' % (p.parse(s),)
diff --git a/freebase/uritemplate.py b/freebase/uritemplate.py
deleted file mode 100755
index 0a81147..0000000
--- a/freebase/uritemplate.py
+++ /dev/null
@@ -1,143 +0,0 @@
-# ========================================================================
-# Copyright (c) 2007, Metaweb Technologies, Inc.
-# All rights reserved.
-#
-# Redistribution and use in source and binary forms, with or without
-# modification, are permitted provided that the following conditions
-# are met:
-# * Redistributions of source code must retain the above copyright
-# notice, this list of conditions and the following disclaimer.
-# * Redistributions in binary form must reproduce the above
-# copyright notice, this list of conditions and the following
-# disclaimer in the documentation and/or other materials provided
-# with the distribution.
-#
-# THIS SOFTWARE IS PROVIDED BY METAWEB TECHNOLOGIES AND CONTRIBUTORS
-# ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
-# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
-# FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL METAWEB
-# TECHNOLOGIES OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
-# INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
-# BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
-# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
-# CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
-# LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
-# ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
-# POSSIBILITY OF SUCH DAMAGE.
-# ========================================================================
-
-#
-# URI Templating in Python
-#
-# see http://bitworking.org/projects/URI-Templates/
-# and http://bitworking.org/news/URI_Templates
-#
-# note that this implementation may go away soon in
-# favor of joe gregorio's own code:
-# http://code.google.com/p/uri-templates/
-#
-#
-# this implementation can also parse URIs, as long as the
-# template is sufficiently specific. to allow this to work
-# the '/' character is forbidden in keys when parsing.
-# later it should be possible to loosen this restriction.
-#
-#
-# example:
-# from whatever.uritemplate import expand_uri_template
-# expand_uri_template('http://{host}/{file}',
-# dict(host='example.org',
-# file='fred'))
-#
-# TODO:
-# allow parsing to be aware of http://www. and trailing /
-# nail down quoting issues
-#
-
-import os, sys, re
-import urllib
-
-__all__ = ['expand_uri_template', 'URITemplate']
-
-
-def expand_uri_template(template, args):
- """Expand a URI template using the given args dictionary.
- """
- return URITemplate(template).run(args)
-
-def _uri_encode_var(v):
- return urllib.quote(v, safe="-_.~!$&'()*+,;=:/?[]#@")
-
-
-class URITemplate(object):
- """a URITemplate is a URI with simple variable substitution.
- """
-
- VARREF = re.compile(r'\{([0-9a-zA-Z_]+)\}')
-
- def __init__(self, s):
- """Compile a URITemplate from a string.
- """
- self.template = s;
-
- self.params = []
- tsplit = self.VARREF.split(s)
- rxs = ['^']
- for i in range(len(tsplit)):
- if i % 2:
- # track the vars used
- self.params.append(tsplit[i])
- # vars match any string
- # vars are assumed to lack '/' - this is imperfect...
- rxs.append('([^/]*)')
- else:
- # quote special chars regexp interpretation
- rxs.append(re.escape(tsplit[i]))
- rxs.append('$')
- self._parser = re.compile(''.join(rxs))
-
- def __repr__(self):
- return '<URITemplate %r>' % self.template
-
- def run (self, args):
- """Expand the template using the given args.
- """
- def repl(m):
- key = m.group(1)
- return _uri_encode_var(args.get(key, ''))
- uri = self.VARREF.sub(repl,self.template)
-
- #if self.parse(uri) is None:
- # print 're-parsing generated uri failed: %r, %r' % (uri, self.template)
- return uri
-
-
- def parse(self, uri):
- """Try to parse a URI, extracting variable values.
- """
- m = self._parser.match(uri)
- if m is None:
- return None
- return dict(zip(self.params, m.groups()))
-
-
-if __name__ == '__main__':
- #
- # testcases are imported from the URI::Template module on CPAN
- #
- import urllib2, simplejson
- fp = urllib2.urlopen('http://search.cpan.org/src/BRICAS/URI-Template-0.09/t/data/spec.json')
- testsuite = simplejson.loads(fp.read())
- vars = dict([(k.encode('utf-8'),v) for k,v in testsuite['variables'].items()])
- nsucceed = 0
- nfail = 0
- for test in testsuite['tests']:
- ut = URITemplate(test['template'])
- uri = ut.run(vars)
- if uri != test['expected']:
- print 'FAILED %r expected %r' % (uri, test['expected'])
- print ' vars: %r' % (vars,)
- nfail += 1
- else:
- nsucceed += 1
- print 'tests: %d succeeded, %d failed' % (nsucceed, nfail)