Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 5 additions & 8 deletions .travis.yml
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,17 +7,14 @@ matrix:
# One build pulls latest versions dynamically
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5 PRESTO=RELEASE SQLALCHEMY=sqlalchemy
# Others use pinned versions.
- python: 3.6
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.5
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 3.4
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.0.12
# stale stuff we're still using / supporting
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
- python: 2.7
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==0.8.7
# exclude: python 3 against old libries
env: CDH=cdh5 CDH_VERSION=5.10.1 PRESTO=0.147 SQLALCHEMY=sqlalchemy==1.2.8
install:
- ./scripts/travis-install.sh
- pip install codecov
Expand Down
11 changes: 8 additions & 3 deletions README.rst
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
.. image:: https://travis-ci.org/dropbox/PyHive.svg?branch=master
:target: https://travis-ci.org/dropbox/PyHive
.. image:: https://img.shields.io/codecov/c/github/dropbox/PyHive.svg

======
PyHive
Expand DownExpand Up@@ -100,9 +101,6 @@ PyHive works with
- Python 2.7 / Python 3
- For Presto: Presto install
- For Hive: `HiveServer2 <https://cwiki.apache.org/confluence/display/Hive/Setting+up+HiveServer2>`_ daemon
- For Python 3 + Hive + SASL, you currently need to install an unreleased version of ``thrift_sasl``
(``pip install git+https://github.com/cloudera/thrift_sasl``).
At the time of writing, the latest version of ``thrift_sasl`` was 0.2.1.

Changelog
=========
Expand DownExpand Up@@ -137,3 +135,10 @@ Run the following in an environment with Hive/Presto::

WARNING: This drops/creates tables named ``one_row``, ``one_row_complex``, and ``many_rows``, plus a
database called ``pyhive_test_database``.

Updating TCLIService
====================

The TCLIService module is autogenerated using a ``TCLIService.thrift`` file. To update it, the
``generate.py`` file can be used: ``python generate.py <TCLIServiceURL>``. When left blank, the
version for Hive 2.3 will be downloaded.
1 change: 0 additions & 1 deletion dev_requirements.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -11,7 +11,6 @@ pytest-timeout==1.2.0
# actual dependencies: let things break if a package changes
requests>=1.0.0
sasl>=0.2.1
sqlalchemy>=0.8.7
thrift>=0.10.0
#thrift_sasl>=0.1.0
git+https://github.com/cloudera/thrift_sasl # Using master branch in order to get Python 3 SASL patches
52 changes: 52 additions & 0 deletions generate.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
"""
This file can be used to generate a new version of the TCLIService package
using a TCLIService.thrift URL.

If no URL is specified, the file for Hive 2.3 will be downloaded.

Usage:

python generate.py THRIFT_URL

or

python generate.py
"""
import shutil
import sys
from os import path
from urllib.request import urlopen
import subprocess

here = path.abspath(path.dirname(__file__))

PACKAGE = 'TCLIService'
GENERATED = 'gen-py'

HIVE_SERVER2_URL = \
'https://raw.githubusercontent.com/apache/hive/branch-2.3/service-rpc/if/TCLIService.thrift'


def save_url(url):
data = urlopen(url).read()
file_path = path.join(here, url.rsplit('/', 1)[-1])
with open(file_path, 'wb') as f:
f.write(data)


def main(hive_server2_url):
save_url(hive_server2_url)
hive_server2_path = path.join(here, hive_server2_url.rsplit('/', 1)[-1])

subprocess.call(['thrift', '-r', '--gen', 'py', hive_server2_path])
shutil.move(path.join(here, PACKAGE), path.join(here, PACKAGE + '.old'))
shutil.move(path.join(here, GENERATED, PACKAGE), path.join(here, PACKAGE))
shutil.rmtree(path.join(here, PACKAGE + '.old'))


if __name__ == '__main__':
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = HIVE_SERVER2_URL
main(url)
20 changes: 3 additions & 17 deletions pyhive/common.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,14 +8,14 @@
from builtins import bytes
from builtins import int
from builtins import object
from builtins import range
from builtins import str
from past.builtins import basestring
from pyhive import exc
import abc
import collections
import time
from future.utils import with_metaclass
from itertools import islice


class DBAPICursor(with_metaclass(abc.ABCMeta, object)):
Expand DownExpand Up@@ -124,14 +124,7 @@ def fetchmany(self, size=None):
"""
if size is None:
size = self.arraysize
result = []
for _ in range(size):
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(islice(iter(self.fetchone, None), size))

def fetchall(self):
"""Fetch all (remaining) rows of a query result, returning them as a sequence of sequences
Expand All@@ -140,14 +133,7 @@ def fetchall(self):
An :py:class:`~pyhive.exc.Error` (or subclass) exception is raised if the previous call to
:py:meth:`execute` did not produce any result set or no call was issued yet.
"""
result = []
while True:
one = self.fetchone()
if one is None:
break
else:
result.append(one)
return result
return list(iter(self.fetchone, None))

@property
def arraysize(self):
Expand Down
68 changes: 65 additions & 3 deletions pyhive/hive.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,11 @@

from __future__ import absolute_import
from __future__ import unicode_literals

import datetime
import re
from decimal import Decimal

from TCLIService import TCLIService
from TCLIService import constants
from TCLIService import ttypes
Expand All@@ -31,6 +36,31 @@

_logger = logging.getLogger(__name__)

_TIMESTAMP_PATTERN = re.compile(r'(\d+-\d+-\d+ \d+:\d+:\d+(\.\d{,6})?)')


def _parse_timestamp(value):
if value:
match = _TIMESTAMP_PATTERN.match(value)
if match:
if match.group(2):
format = '%Y-%m-%d %H:%M:%S.%f'
# use the pattern to truncate the value
value = match.group()
else:
format = '%Y-%m-%d %H:%M:%S'
value = datetime.datetime.strptime(value, format)
else:
raise Exception(
'Cannot convert "{}" into a datetime'.format(value))
else:
value = None
return value


TYPES_CONVERTER = {"DECIMAL_TYPE": Decimal,
"TIMESTAMP_TYPE": _parse_timestamp}


class HiveParamEscaper(common.ParamEscaper):
def escape_string(self, item):
Expand DownExpand Up@@ -177,6 +207,14 @@ def sasl_factory():
self._transport.close()
raise

def __enter__(self):
"""Transport should already be opened by __init__"""
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Call close"""
self.close()

def close(self):
"""Close the underlying session and Thrift transport"""
req = ttypes.TCloseSessionReq(sessionHandle=self._sessionHandle)
Expand DownExpand Up@@ -215,7 +253,7 @@ class Cursor(common.DBAPICursor):
def __init__(self, connection, arraysize=1000):
self._operationHandle = None
super(Cursor, self).__init__()
self.arraysize = arraysize
self._arraysize = arraysize
self._connection = connection

def _reset_state(self):
Expand All@@ -230,6 +268,19 @@ def _reset_state(self):
finally:
self._operationHandle = None

@property
def arraysize(self):
return self._arraysize

@arraysize.setter
def arraysize(self, value):
"""Array size cannot be None, and should be an integer"""
default_arraysize = 1000
try:
self._arraysize = int(value) or default_arraysize
except TypeError:
self._arraysize = default_arraysize

@property
def description(self):
"""This read-only attribute is a sequence of 7-item sequences.
Expand DownExpand Up@@ -273,6 +324,12 @@ def description(self):
))
return self._description

def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
self.close()

def close(self):
"""Close the operation handle"""
self._reset_state()
Expand DownExpand Up@@ -320,8 +377,10 @@ def _fetch_more(self):
)
response = self._connection.client.FetchResults(req)
_check_status(response)
schema = self.description
assert not response.results.rows, 'expected data in columnar format'
columns = map(_unwrap_column, response.results.columns)
columns = [_unwrap_column(col, col_schema[1]) for col, col_schema in
zip(response.results.columns, schema)]
new_data = list(zip(*columns))
self._data += new_data
# response.hasMoreRows seems to always be False, so we instead check the number of rows
Expand DownExpand Up@@ -402,7 +461,7 @@ def fetch_logs(self):
#


def _unwrap_column(col):
def _unwrap_column(col, type_=None):
"""Return a list of raw values from a TColumn instance."""
for attr, wrapper in iteritems(col.__dict__):
if wrapper is not None:
Expand All@@ -414,6 +473,9 @@ def _unwrap_column(col):
for b in range(8):
if byte & (1 << b):
result[i * 8 + b] = None
converter = TYPES_CONVERTER.get(type_, None)
if converter and type_:
result = [converter(row) if row else row for row in result]
return result
raise DataError("Got empty column value {}".format(col)) # pragma: no cover

Expand Down
Loading