mirror of https://github.com/apache/cassandra
cqlsh: handle unicode BOM at the start of files
Patch by Abhishek Gupta; reviewed by Tyler Hobbs for CASSANDRA-8638
This commit is contained in:
parent
1e5a0e19a9
commit
c49f6666e4
|
|
@ -1,4 +1,5 @@
|
|||
2.1.3
|
||||
* (cqlsh) Handle unicode BOM at start of files (CASSANDRA-8638)
|
||||
* Stop compactions before exiting offline tools (CASSANDRA-8623)
|
||||
* Update tools/stress/README.txt to match current behaviour (CASSANDRA-7933)
|
||||
* Fix schema from Thrift conversion with empty metadata (CASSANDRA-8695)
|
||||
|
|
|
|||
10
bin/cqlsh
10
bin/cqlsh
|
|
@ -121,7 +121,7 @@ from cqlshlib import cqlhandling, cql3handling, pylexotron, sslhandling, async_i
|
|||
from cqlshlib.displaying import (RED, BLUE, CYAN, ANSI_RESET, COLUMN_NAME_COLORS,
|
||||
FormattedValue, colorme)
|
||||
from cqlshlib.formatting import format_by_type, formatter_for, format_value_utype
|
||||
from cqlshlib.util import trim_if_present
|
||||
from cqlshlib.util import trim_if_present, get_file_encoding_bomsize
|
||||
from cqlshlib.tracing import print_trace_session, print_trace
|
||||
|
||||
DEFAULT_HOST = '127.0.0.1'
|
||||
|
|
@ -1601,7 +1601,9 @@ class Shell(cmd.Cmd):
|
|||
fname = parsed.get_binding('fname')
|
||||
fname = os.path.expanduser(self.cql_unprotect_value(fname))
|
||||
try:
|
||||
f = open(fname, 'r')
|
||||
encoding, bom_size = get_file_encoding_bomsize(fname)
|
||||
f = codecs.open(fname, 'r', encoding)
|
||||
f.seek(bom_size)
|
||||
except IOError, e:
|
||||
self.printerr('Could not open %r: %s' % (fname, e))
|
||||
return
|
||||
|
|
@ -2013,7 +2015,9 @@ def main(options, hostname, port):
|
|||
stdin = None
|
||||
else:
|
||||
try:
|
||||
stdin = open(options.file, 'r')
|
||||
encoding, bom_size = get_file_encoding_bomsize(options.file)
|
||||
stdin = codecs.open(options.file, 'r', encoding)
|
||||
stdin.seek(bom_size)
|
||||
except IOError, e:
|
||||
sys.exit("Can't open %r: %s" % (options.file, e))
|
||||
|
||||
|
|
|
|||
|
|
@ -14,8 +14,10 @@
|
|||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import codecs
|
||||
from itertools import izip
|
||||
|
||||
|
||||
def split_list(items, pred):
|
||||
"""
|
||||
Split up a list (or other iterable) on the elements which satisfy the
|
||||
|
|
@ -36,6 +38,7 @@ def split_list(items, pred):
|
|||
results.append(thisresult)
|
||||
return results
|
||||
|
||||
|
||||
def find_common_prefix(strs):
|
||||
"""
|
||||
Given a list (iterable) of strings, return the longest common prefix.
|
||||
|
|
@ -54,6 +57,7 @@ def find_common_prefix(strs):
|
|||
break
|
||||
return ''.join(common)
|
||||
|
||||
|
||||
def list_bifilter(pred, iterable):
|
||||
"""
|
||||
Filter an iterable into two output lists: the first containing all
|
||||
|
|
@ -70,10 +74,35 @@ def list_bifilter(pred, iterable):
|
|||
(yes_s if pred(i) else no_s).append(i)
|
||||
return yes_s, no_s
|
||||
|
||||
|
||||
def identity(x):
|
||||
return x
|
||||
|
||||
|
||||
def trim_if_present(s, prefix):
|
||||
if s.startswith(prefix):
|
||||
return s[len(prefix):]
|
||||
return s
|
||||
|
||||
|
||||
def get_file_encoding_bomsize(filename):
|
||||
"""
|
||||
Checks the beginning of a file for a Unicode BOM. Based on this check,
|
||||
the encoding that should be used to open the file and the number of
|
||||
bytes that should be skipped (to skip the BOM) are returned.
|
||||
"""
|
||||
bom_encodings = ((codecs.BOM_UTF8, 'utf-8-sig'),
|
||||
(codecs.BOM_UTF16_LE, 'utf-16le'),
|
||||
(codecs.BOM_UTF16_BE, 'utf-16be'),
|
||||
(codecs.BOM_UTF32_LE, 'utf-32be'),
|
||||
(codecs.BOM_UTF32_BE, 'utf-32be'))
|
||||
|
||||
firstbytes = open(filename, 'rb').read(4)
|
||||
for bom, encoding in bom_encodings:
|
||||
if firstbytes.startswith(bom):
|
||||
file_encoding, size = encoding, len(bom)
|
||||
break
|
||||
else:
|
||||
file_encoding, size = "ascii", 0
|
||||
|
||||
return (file_encoding, size)
|
||||
|
|
|
|||
Loading…
Reference in New Issue