init
This commit is contained in:
@@ -0,0 +1,3 @@
|
||||
from kafka.record.memory_records import MemoryRecords, MemoryRecordsBuilder
|
||||
|
||||
__all__ = ["MemoryRecords", "MemoryRecordsBuilder"]
|
||||
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env python
|
||||
#
|
||||
# Taken from https://cloud.google.com/appengine/docs/standard/python/refdocs/\
|
||||
# modules/google/appengine/api/files/crc32c?hl=ru
|
||||
#
|
||||
# Copyright 2007 Google Inc.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
"""Implementation of CRC-32C checksumming as in rfc3720 section B.4.
|
||||
See https://en.wikipedia.org/wiki/Cyclic_redundancy_check for details on CRC-32C
|
||||
This code is a manual python translation of c code generated by
|
||||
pycrc 0.7.1 (https://pycrc.org/). Command line used:
|
||||
'./pycrc.py --model=crc-32c --generate c --algorithm=table-driven'
|
||||
"""
|
||||
|
||||
import array
|
||||
|
||||
CRC_TABLE = (
|
||||
0x00000000, 0xf26b8303, 0xe13b70f7, 0x1350f3f4,
|
||||
0xc79a971f, 0x35f1141c, 0x26a1e7e8, 0xd4ca64eb,
|
||||
0x8ad958cf, 0x78b2dbcc, 0x6be22838, 0x9989ab3b,
|
||||
0x4d43cfd0, 0xbf284cd3, 0xac78bf27, 0x5e133c24,
|
||||
0x105ec76f, 0xe235446c, 0xf165b798, 0x030e349b,
|
||||
0xd7c45070, 0x25afd373, 0x36ff2087, 0xc494a384,
|
||||
0x9a879fa0, 0x68ec1ca3, 0x7bbcef57, 0x89d76c54,
|
||||
0x5d1d08bf, 0xaf768bbc, 0xbc267848, 0x4e4dfb4b,
|
||||
0x20bd8ede, 0xd2d60ddd, 0xc186fe29, 0x33ed7d2a,
|
||||
0xe72719c1, 0x154c9ac2, 0x061c6936, 0xf477ea35,
|
||||
0xaa64d611, 0x580f5512, 0x4b5fa6e6, 0xb93425e5,
|
||||
0x6dfe410e, 0x9f95c20d, 0x8cc531f9, 0x7eaeb2fa,
|
||||
0x30e349b1, 0xc288cab2, 0xd1d83946, 0x23b3ba45,
|
||||
0xf779deae, 0x05125dad, 0x1642ae59, 0xe4292d5a,
|
||||
0xba3a117e, 0x4851927d, 0x5b016189, 0xa96ae28a,
|
||||
0x7da08661, 0x8fcb0562, 0x9c9bf696, 0x6ef07595,
|
||||
0x417b1dbc, 0xb3109ebf, 0xa0406d4b, 0x522bee48,
|
||||
0x86e18aa3, 0x748a09a0, 0x67dafa54, 0x95b17957,
|
||||
0xcba24573, 0x39c9c670, 0x2a993584, 0xd8f2b687,
|
||||
0x0c38d26c, 0xfe53516f, 0xed03a29b, 0x1f682198,
|
||||
0x5125dad3, 0xa34e59d0, 0xb01eaa24, 0x42752927,
|
||||
0x96bf4dcc, 0x64d4cecf, 0x77843d3b, 0x85efbe38,
|
||||
0xdbfc821c, 0x2997011f, 0x3ac7f2eb, 0xc8ac71e8,
|
||||
0x1c661503, 0xee0d9600, 0xfd5d65f4, 0x0f36e6f7,
|
||||
0x61c69362, 0x93ad1061, 0x80fde395, 0x72966096,
|
||||
0xa65c047d, 0x5437877e, 0x4767748a, 0xb50cf789,
|
||||
0xeb1fcbad, 0x197448ae, 0x0a24bb5a, 0xf84f3859,
|
||||
0x2c855cb2, 0xdeeedfb1, 0xcdbe2c45, 0x3fd5af46,
|
||||
0x7198540d, 0x83f3d70e, 0x90a324fa, 0x62c8a7f9,
|
||||
0xb602c312, 0x44694011, 0x5739b3e5, 0xa55230e6,
|
||||
0xfb410cc2, 0x092a8fc1, 0x1a7a7c35, 0xe811ff36,
|
||||
0x3cdb9bdd, 0xceb018de, 0xdde0eb2a, 0x2f8b6829,
|
||||
0x82f63b78, 0x709db87b, 0x63cd4b8f, 0x91a6c88c,
|
||||
0x456cac67, 0xb7072f64, 0xa457dc90, 0x563c5f93,
|
||||
0x082f63b7, 0xfa44e0b4, 0xe9141340, 0x1b7f9043,
|
||||
0xcfb5f4a8, 0x3dde77ab, 0x2e8e845f, 0xdce5075c,
|
||||
0x92a8fc17, 0x60c37f14, 0x73938ce0, 0x81f80fe3,
|
||||
0x55326b08, 0xa759e80b, 0xb4091bff, 0x466298fc,
|
||||
0x1871a4d8, 0xea1a27db, 0xf94ad42f, 0x0b21572c,
|
||||
0xdfeb33c7, 0x2d80b0c4, 0x3ed04330, 0xccbbc033,
|
||||
0xa24bb5a6, 0x502036a5, 0x4370c551, 0xb11b4652,
|
||||
0x65d122b9, 0x97baa1ba, 0x84ea524e, 0x7681d14d,
|
||||
0x2892ed69, 0xdaf96e6a, 0xc9a99d9e, 0x3bc21e9d,
|
||||
0xef087a76, 0x1d63f975, 0x0e330a81, 0xfc588982,
|
||||
0xb21572c9, 0x407ef1ca, 0x532e023e, 0xa145813d,
|
||||
0x758fe5d6, 0x87e466d5, 0x94b49521, 0x66df1622,
|
||||
0x38cc2a06, 0xcaa7a905, 0xd9f75af1, 0x2b9cd9f2,
|
||||
0xff56bd19, 0x0d3d3e1a, 0x1e6dcdee, 0xec064eed,
|
||||
0xc38d26c4, 0x31e6a5c7, 0x22b65633, 0xd0ddd530,
|
||||
0x0417b1db, 0xf67c32d8, 0xe52cc12c, 0x1747422f,
|
||||
0x49547e0b, 0xbb3ffd08, 0xa86f0efc, 0x5a048dff,
|
||||
0x8ecee914, 0x7ca56a17, 0x6ff599e3, 0x9d9e1ae0,
|
||||
0xd3d3e1ab, 0x21b862a8, 0x32e8915c, 0xc083125f,
|
||||
0x144976b4, 0xe622f5b7, 0xf5720643, 0x07198540,
|
||||
0x590ab964, 0xab613a67, 0xb831c993, 0x4a5a4a90,
|
||||
0x9e902e7b, 0x6cfbad78, 0x7fab5e8c, 0x8dc0dd8f,
|
||||
0xe330a81a, 0x115b2b19, 0x020bd8ed, 0xf0605bee,
|
||||
0x24aa3f05, 0xd6c1bc06, 0xc5914ff2, 0x37faccf1,
|
||||
0x69e9f0d5, 0x9b8273d6, 0x88d28022, 0x7ab90321,
|
||||
0xae7367ca, 0x5c18e4c9, 0x4f48173d, 0xbd23943e,
|
||||
0xf36e6f75, 0x0105ec76, 0x12551f82, 0xe03e9c81,
|
||||
0x34f4f86a, 0xc69f7b69, 0xd5cf889d, 0x27a40b9e,
|
||||
0x79b737ba, 0x8bdcb4b9, 0x988c474d, 0x6ae7c44e,
|
||||
0xbe2da0a5, 0x4c4623a6, 0x5f16d052, 0xad7d5351,
|
||||
)
|
||||
|
||||
CRC_INIT = 0
|
||||
_MASK = 0xFFFFFFFF
|
||||
|
||||
|
||||
def crc_update(crc, data):
|
||||
"""Update CRC-32C checksum with data.
|
||||
Args:
|
||||
crc: 32-bit checksum to update as long.
|
||||
data: byte array, string or iterable over bytes.
|
||||
Returns:
|
||||
32-bit updated CRC-32C as long.
|
||||
"""
|
||||
if not isinstance(data, array.array) or data.itemsize != 1:
|
||||
buf = array.array("B", data)
|
||||
else:
|
||||
buf = data
|
||||
crc = crc ^ _MASK
|
||||
for b in buf:
|
||||
table_index = (crc ^ b) & 0xff
|
||||
crc = (CRC_TABLE[table_index] ^ (crc >> 8)) & _MASK
|
||||
return crc ^ _MASK
|
||||
|
||||
|
||||
def crc_finalize(crc):
|
||||
"""Finalize CRC-32C checksum.
|
||||
This function should be called as last step of crc calculation.
|
||||
Args:
|
||||
crc: 32-bit checksum as long.
|
||||
Returns:
|
||||
finalized 32-bit checksum as long
|
||||
"""
|
||||
return crc & _MASK
|
||||
|
||||
|
||||
def crc(data):
|
||||
"""Compute CRC-32C checksum of the data.
|
||||
Args:
|
||||
data: byte array, string or iterable over bytes.
|
||||
Returns:
|
||||
32-bit CRC-32C checksum of data as long.
|
||||
"""
|
||||
return crc_finalize(crc_update(CRC_INIT, data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
# TODO remove the pylint disable once pylint fixes
|
||||
# https://github.com/PyCQA/pylint/issues/2571
|
||||
data = sys.stdin.read() # pylint: disable=assignment-from-no-return
|
||||
print(hex(crc(data)))
|
||||
@@ -0,0 +1,152 @@
|
||||
from __future__ import absolute_import
|
||||
|
||||
import abc
|
||||
|
||||
from kafka.vendor.six import add_metaclass
|
||||
|
||||
|
||||
@add_metaclass(abc.ABCMeta)
|
||||
class ABCRecord(object):
|
||||
__slots__ = ()
|
||||
|
||||
@abc.abstractproperty
|
||||
def size_in_bytes(self):
|
||||
""" Number of total bytes in record
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def offset(self):
|
||||
""" Absolute offset of record
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def timestamp(self):
|
||||
""" Epoch milliseconds
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def timestamp_type(self):
|
||||
""" CREATE_TIME(0) or APPEND_TIME(1)
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def key(self):
|
||||
""" Bytes key or None
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def value(self):
|
||||
""" Bytes value or None
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def checksum(self):
|
||||
""" Prior to v2 format CRC was contained in every message. This will
|
||||
be the checksum for v0 and v1 and None for v2 and above.
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def validate_crc(self):
|
||||
""" Return True if v0/v1 record matches checksum. noop/True for v2 records
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def headers(self):
|
||||
""" If supported by version list of key-value tuples, or empty list if
|
||||
not supported by format.
|
||||
"""
|
||||
|
||||
|
||||
@add_metaclass(abc.ABCMeta)
|
||||
class ABCRecordBatchBuilder(object):
|
||||
__slots__ = ()
|
||||
|
||||
@abc.abstractmethod
|
||||
def append(self, offset, timestamp, key, value, headers=None):
|
||||
""" Writes record to internal buffer.
|
||||
|
||||
Arguments:
|
||||
offset (int): Relative offset of record, starting from 0
|
||||
timestamp (int or None): Timestamp in milliseconds since beginning
|
||||
of the epoch (midnight Jan 1, 1970 (UTC)). If omitted, will be
|
||||
set to current time.
|
||||
key (bytes or None): Key of the record
|
||||
value (bytes or None): Value of the record
|
||||
headers (List[Tuple[str, bytes]]): Headers of the record. Header
|
||||
keys can not be ``None``.
|
||||
|
||||
Returns:
|
||||
(bytes, int): Checksum of the written record (or None for v2 and
|
||||
above) and size of the written record.
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def size_in_bytes(self, offset, timestamp, key, value, headers):
|
||||
""" Return the expected size change on buffer (uncompressed) if we add
|
||||
this message. This will account for varint size changes and give a
|
||||
reliable size.
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def build(self):
|
||||
""" Close for append, compress if needed, write size and header and
|
||||
return a ready to send buffer object.
|
||||
|
||||
Return:
|
||||
bytearray: finished batch, ready to send.
|
||||
"""
|
||||
|
||||
|
||||
@add_metaclass(abc.ABCMeta)
|
||||
class ABCRecordBatch(object):
|
||||
""" For v2 encapsulates a RecordBatch, for v0/v1 a single (maybe
|
||||
compressed) message.
|
||||
"""
|
||||
__slots__ = ()
|
||||
|
||||
@abc.abstractmethod
|
||||
def __iter__(self):
|
||||
""" Return iterator over records (ABCRecord instances). Will decompress
|
||||
if needed.
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def base_offset(self):
|
||||
""" Return base offset for batch
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def size_in_bytes(self):
|
||||
""" Return size of batch in bytes (includes header overhead)
|
||||
"""
|
||||
|
||||
@abc.abstractproperty
|
||||
def magic(self):
|
||||
""" Return magic value (0, 1, 2) for batch.
|
||||
"""
|
||||
|
||||
|
||||
@add_metaclass(abc.ABCMeta)
|
||||
class ABCRecords(object):
|
||||
__slots__ = ()
|
||||
|
||||
@abc.abstractmethod
|
||||
def __init__(self, buffer):
|
||||
""" Initialize with bytes-like object conforming to the buffer
|
||||
interface (ie. bytes, bytearray, memoryview etc.).
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def size_in_bytes(self):
|
||||
""" Returns the size of inner buffer.
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def next_batch(self):
|
||||
""" Return next batch of records (ABCRecordBatch instances).
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def has_next(self):
|
||||
""" True if there are more batches to read, False otherwise.
|
||||
"""
|
||||
@@ -0,0 +1,776 @@
|
||||
# See:
|
||||
# https://github.com/apache/kafka/blob/trunk/clients/src/main/java/org/\
|
||||
# apache/kafka/common/record/DefaultRecordBatch.java
|
||||
# https://github.com/apache/kafka/blob/trunk/clients/src/main/java/org/\
|
||||
# apache/kafka/common/record/DefaultRecord.java
|
||||
|
||||
# RecordBatch and Record implementation for magic 2 and above.
|
||||
# The schema is given below:
|
||||
|
||||
# RecordBatch =>
|
||||
# BaseOffset => Int64
|
||||
# Length => Int32
|
||||
# PartitionLeaderEpoch => Int32
|
||||
# Magic => Int8
|
||||
# CRC => Uint32
|
||||
# Attributes => Int16
|
||||
# LastOffsetDelta => Int32 // also serves as LastSequenceDelta
|
||||
# FirstTimestamp => Int64
|
||||
# MaxTimestamp => Int64
|
||||
# ProducerId => Int64
|
||||
# ProducerEpoch => Int16
|
||||
# BaseSequence => Int32
|
||||
# Records => [Record]
|
||||
|
||||
# Record =>
|
||||
# Length => Varint
|
||||
# Attributes => Int8
|
||||
# TimestampDelta => Varlong
|
||||
# OffsetDelta => Varint
|
||||
# Key => Bytes
|
||||
# Value => Bytes
|
||||
# Headers => [HeaderKey HeaderValue]
|
||||
# HeaderKey => String
|
||||
# HeaderValue => Bytes
|
||||
|
||||
# Note that when compression is enabled (see attributes below), the compressed
|
||||
# record data is serialized directly following the count of the number of
|
||||
# records. (ie Records => [Record], but without length bytes)
|
||||
|
||||
# The CRC covers the data from the attributes to the end of the batch (i.e. all
|
||||
# the bytes that follow the CRC). It is located after the magic byte, which
|
||||
# means that clients must parse the magic byte before deciding how to interpret
|
||||
# the bytes between the batch length and the magic byte. The partition leader
|
||||
# epoch field is not included in the CRC computation to avoid the need to
|
||||
# recompute the CRC when this field is assigned for every batch that is
|
||||
# received by the broker. The CRC-32C (Castagnoli) polynomial is used for the
|
||||
# computation.
|
||||
|
||||
# The current RecordBatch attributes are given below:
|
||||
#
|
||||
# * Unused (6-15)
|
||||
# * Control (5)
|
||||
# * Transactional (4)
|
||||
# * Timestamp Type (3)
|
||||
# * Compression Type (0-2)
|
||||
|
||||
import struct
|
||||
import time
|
||||
from kafka.record.abc import ABCRecord, ABCRecordBatch, ABCRecordBatchBuilder
|
||||
from kafka.record.util import (
|
||||
decode_varint, encode_varint, calc_crc32c, size_of_varint
|
||||
)
|
||||
from kafka.errors import CorruptRecordError, UnsupportedCodecError
|
||||
from kafka.codec import (
|
||||
gzip_encode, snappy_encode, lz4_encode, zstd_encode,
|
||||
gzip_decode, snappy_decode, lz4_decode, zstd_decode
|
||||
)
|
||||
import kafka.codec as codecs
|
||||
|
||||
|
||||
class DefaultRecordBase(object):
|
||||
|
||||
__slots__ = ()
|
||||
|
||||
HEADER_STRUCT = struct.Struct(
|
||||
">q" # BaseOffset => Int64
|
||||
"i" # Length => Int32
|
||||
"i" # PartitionLeaderEpoch => Int32
|
||||
"b" # Magic => Int8
|
||||
"I" # CRC => Uint32
|
||||
"h" # Attributes => Int16
|
||||
"i" # LastOffsetDelta => Int32 // also serves as LastSequenceDelta
|
||||
"q" # FirstTimestamp => Int64
|
||||
"q" # MaxTimestamp => Int64
|
||||
"q" # ProducerId => Int64
|
||||
"h" # ProducerEpoch => Int16
|
||||
"i" # BaseSequence => Int32
|
||||
"i" # Records count => Int32
|
||||
)
|
||||
# Byte offset in HEADER_STRUCT of attributes field. Used to calculate CRC
|
||||
ATTRIBUTES_OFFSET = struct.calcsize(">qiibI")
|
||||
CRC_OFFSET = struct.calcsize(">qiib")
|
||||
AFTER_LEN_OFFSET = struct.calcsize(">qi")
|
||||
|
||||
CODEC_MASK = 0x07
|
||||
CODEC_NONE = 0x00
|
||||
CODEC_GZIP = 0x01
|
||||
CODEC_SNAPPY = 0x02
|
||||
CODEC_LZ4 = 0x03
|
||||
CODEC_ZSTD = 0x04
|
||||
TIMESTAMP_TYPE_MASK = 0x08
|
||||
TRANSACTIONAL_MASK = 0x10
|
||||
CONTROL_MASK = 0x20
|
||||
|
||||
LOG_APPEND_TIME = 1
|
||||
CREATE_TIME = 0
|
||||
NO_PRODUCER_ID = -1
|
||||
NO_SEQUENCE = -1
|
||||
MAX_INT = 2147483647
|
||||
|
||||
def _assert_has_codec(self, compression_type):
|
||||
if compression_type == self.CODEC_GZIP:
|
||||
checker, name = codecs.has_gzip, "gzip"
|
||||
elif compression_type == self.CODEC_SNAPPY:
|
||||
checker, name = codecs.has_snappy, "snappy"
|
||||
elif compression_type == self.CODEC_LZ4:
|
||||
checker, name = codecs.has_lz4, "lz4"
|
||||
elif compression_type == self.CODEC_ZSTD:
|
||||
checker, name = codecs.has_zstd, "zstd"
|
||||
else:
|
||||
raise UnsupportedCodecError("Unrecognized compression type: %s" % (compression_type,))
|
||||
if not checker():
|
||||
raise UnsupportedCodecError(
|
||||
"Libraries for {} compression codec not found".format(name))
|
||||
|
||||
|
||||
class DefaultRecordBatch(DefaultRecordBase, ABCRecordBatch):
|
||||
|
||||
__slots__ = ("_buffer", "_header_data", "_pos", "_num_records",
|
||||
"_next_record_index", "_decompressed")
|
||||
|
||||
def __init__(self, buffer):
|
||||
self._buffer = bytearray(buffer)
|
||||
self._header_data = self.HEADER_STRUCT.unpack_from(self._buffer)
|
||||
self._pos = self.HEADER_STRUCT.size
|
||||
self._num_records = self._header_data[12]
|
||||
self._next_record_index = 0
|
||||
self._decompressed = False
|
||||
|
||||
@property
|
||||
def base_offset(self):
|
||||
return self._header_data[0]
|
||||
|
||||
@property
|
||||
def size_in_bytes(self):
|
||||
return self._header_data[1] + self.AFTER_LEN_OFFSET
|
||||
|
||||
@property
|
||||
def leader_epoch(self):
|
||||
return self._header_data[2]
|
||||
|
||||
@property
|
||||
def magic(self):
|
||||
return self._header_data[3]
|
||||
|
||||
@property
|
||||
def crc(self):
|
||||
return self._header_data[4]
|
||||
|
||||
@property
|
||||
def attributes(self):
|
||||
return self._header_data[5]
|
||||
|
||||
@property
|
||||
def last_offset_delta(self):
|
||||
return self._header_data[6]
|
||||
|
||||
@property
|
||||
def last_offset(self):
|
||||
return self.base_offset + self.last_offset_delta
|
||||
|
||||
@property
|
||||
def next_offset(self):
|
||||
return self.last_offset + 1
|
||||
|
||||
@property
|
||||
def compression_type(self):
|
||||
return self.attributes & self.CODEC_MASK
|
||||
|
||||
@property
|
||||
def timestamp_type(self):
|
||||
return int(bool(self.attributes & self.TIMESTAMP_TYPE_MASK))
|
||||
|
||||
@property
|
||||
def is_transactional(self):
|
||||
return bool(self.attributes & self.TRANSACTIONAL_MASK)
|
||||
|
||||
@property
|
||||
def is_control_batch(self):
|
||||
return bool(self.attributes & self.CONTROL_MASK)
|
||||
|
||||
@property
|
||||
def first_timestamp(self):
|
||||
return self._header_data[7]
|
||||
|
||||
@property
|
||||
def max_timestamp(self):
|
||||
return self._header_data[8]
|
||||
|
||||
@property
|
||||
def producer_id(self):
|
||||
return self._header_data[9]
|
||||
|
||||
def has_producer_id(self):
|
||||
return self.producer_id > self.NO_PRODUCER_ID
|
||||
|
||||
@property
|
||||
def producer_epoch(self):
|
||||
return self._header_data[10]
|
||||
|
||||
@property
|
||||
def base_sequence(self):
|
||||
return self._header_data[11]
|
||||
|
||||
@property
|
||||
def has_sequence(self):
|
||||
return self._header_data[11] != -1 # NO_SEQUENCE
|
||||
|
||||
@property
|
||||
def last_sequence(self):
|
||||
if self.base_sequence == self.NO_SEQUENCE:
|
||||
return self.NO_SEQUENCE
|
||||
return self._increment_sequence(self.base_sequence, self.last_offset_delta)
|
||||
|
||||
def _increment_sequence(self, base, increment):
|
||||
if base > (self.MAX_INT - increment):
|
||||
return increment - (self.MAX_INT - base) - 1
|
||||
return base + increment
|
||||
|
||||
@property
|
||||
def records_count(self):
|
||||
return self._header_data[12]
|
||||
|
||||
def _maybe_uncompress(self):
|
||||
if not self._decompressed:
|
||||
compression_type = self.compression_type
|
||||
if compression_type != self.CODEC_NONE:
|
||||
self._assert_has_codec(compression_type)
|
||||
data = memoryview(self._buffer)[self._pos:]
|
||||
if compression_type == self.CODEC_GZIP:
|
||||
uncompressed = gzip_decode(data)
|
||||
if compression_type == self.CODEC_SNAPPY:
|
||||
uncompressed = snappy_decode(data.tobytes())
|
||||
if compression_type == self.CODEC_LZ4:
|
||||
uncompressed = lz4_decode(data.tobytes())
|
||||
if compression_type == self.CODEC_ZSTD:
|
||||
uncompressed = zstd_decode(data.tobytes())
|
||||
self._buffer = bytearray(uncompressed)
|
||||
self._pos = 0
|
||||
self._decompressed = True
|
||||
|
||||
def _read_msg(
|
||||
self,
|
||||
decode_varint=decode_varint):
|
||||
# Record =>
|
||||
# Length => Varint
|
||||
# Attributes => Int8
|
||||
# TimestampDelta => Varlong
|
||||
# OffsetDelta => Varint
|
||||
# Key => Bytes
|
||||
# Value => Bytes
|
||||
# Headers => [HeaderKey HeaderValue]
|
||||
# HeaderKey => String
|
||||
# HeaderValue => Bytes
|
||||
|
||||
buffer = self._buffer
|
||||
pos = self._pos
|
||||
length, pos = decode_varint(buffer, pos)
|
||||
start_pos = pos
|
||||
_, pos = decode_varint(buffer, pos) # attrs can be skipped for now
|
||||
|
||||
ts_delta, pos = decode_varint(buffer, pos)
|
||||
if self.timestamp_type == self.LOG_APPEND_TIME:
|
||||
timestamp = self.max_timestamp
|
||||
else:
|
||||
timestamp = self.first_timestamp + ts_delta
|
||||
|
||||
offset_delta, pos = decode_varint(buffer, pos)
|
||||
offset = self.base_offset + offset_delta
|
||||
|
||||
key_len, pos = decode_varint(buffer, pos)
|
||||
if key_len >= 0:
|
||||
key = bytes(buffer[pos: pos + key_len])
|
||||
pos += key_len
|
||||
else:
|
||||
key = None
|
||||
|
||||
value_len, pos = decode_varint(buffer, pos)
|
||||
if value_len >= 0:
|
||||
value = bytes(buffer[pos: pos + value_len])
|
||||
pos += value_len
|
||||
else:
|
||||
value = None
|
||||
|
||||
header_count, pos = decode_varint(buffer, pos)
|
||||
if header_count < 0:
|
||||
raise CorruptRecordError("Found invalid number of record "
|
||||
"headers {}".format(header_count))
|
||||
headers = []
|
||||
while header_count:
|
||||
# Header key is of type String, that can't be None
|
||||
h_key_len, pos = decode_varint(buffer, pos)
|
||||
if h_key_len < 0:
|
||||
raise CorruptRecordError(
|
||||
"Invalid negative header key size {}".format(h_key_len))
|
||||
h_key = buffer[pos: pos + h_key_len].decode("utf-8")
|
||||
pos += h_key_len
|
||||
|
||||
# Value is of type NULLABLE_BYTES, so it can be None
|
||||
h_value_len, pos = decode_varint(buffer, pos)
|
||||
if h_value_len >= 0:
|
||||
h_value = bytes(buffer[pos: pos + h_value_len])
|
||||
pos += h_value_len
|
||||
else:
|
||||
h_value = None
|
||||
|
||||
headers.append((h_key, h_value))
|
||||
header_count -= 1
|
||||
|
||||
# validate whether we have read all header bytes in the current record
|
||||
if pos - start_pos != length:
|
||||
raise CorruptRecordError(
|
||||
"Invalid record size: expected to read {} bytes in record "
|
||||
"payload, but instead read {}".format(length, pos - start_pos))
|
||||
self._pos = pos
|
||||
|
||||
if self.is_control_batch:
|
||||
return ControlRecord(
|
||||
length, offset, timestamp, self.timestamp_type, key, value, headers)
|
||||
else:
|
||||
return DefaultRecord(
|
||||
length, offset, timestamp, self.timestamp_type, key, value, headers)
|
||||
|
||||
def __iter__(self):
|
||||
self._maybe_uncompress()
|
||||
return self
|
||||
|
||||
def __next__(self):
|
||||
if self._next_record_index >= self._num_records:
|
||||
if self._pos != len(self._buffer):
|
||||
raise CorruptRecordError(
|
||||
"{} unconsumed bytes after all records consumed".format(
|
||||
len(self._buffer) - self._pos))
|
||||
raise StopIteration
|
||||
try:
|
||||
msg = self._read_msg()
|
||||
except (ValueError, IndexError) as err:
|
||||
raise CorruptRecordError(
|
||||
"Found invalid record structure: {!r}".format(err))
|
||||
else:
|
||||
self._next_record_index += 1
|
||||
return msg
|
||||
|
||||
next = __next__
|
||||
|
||||
def validate_crc(self):
|
||||
assert self._decompressed is False, \
|
||||
"Validate should be called before iteration"
|
||||
|
||||
crc = self.crc
|
||||
data_view = memoryview(self._buffer)[self.ATTRIBUTES_OFFSET:]
|
||||
verify_crc = calc_crc32c(data_view.tobytes())
|
||||
return crc == verify_crc
|
||||
|
||||
def __str__(self):
|
||||
return (
|
||||
"DefaultRecordBatch(magic={}, base_offset={}, last_offset_delta={},"
|
||||
" first_timestamp={}, max_timestamp={},"
|
||||
" is_transactional={}, producer_id={}, producer_epoch={}, base_sequence={},"
|
||||
" records_count={})".format(
|
||||
self.magic, self.base_offset, self.last_offset_delta,
|
||||
self.first_timestamp, self.max_timestamp,
|
||||
self.is_transactional, self.producer_id, self.producer_epoch, self.base_sequence,
|
||||
self.records_count))
|
||||
|
||||
|
||||
class DefaultRecord(ABCRecord):
|
||||
|
||||
__slots__ = ("_size_in_bytes", "_offset", "_timestamp", "_timestamp_type", "_key", "_value",
|
||||
"_headers")
|
||||
|
||||
def __init__(self, size_in_bytes, offset, timestamp, timestamp_type, key, value, headers):
|
||||
self._size_in_bytes = size_in_bytes
|
||||
self._offset = offset
|
||||
self._timestamp = timestamp
|
||||
self._timestamp_type = timestamp_type
|
||||
self._key = key
|
||||
self._value = value
|
||||
self._headers = headers
|
||||
|
||||
@property
|
||||
def size_in_bytes(self):
|
||||
return self._size_in_bytes
|
||||
|
||||
@property
|
||||
def offset(self):
|
||||
return self._offset
|
||||
|
||||
@property
|
||||
def timestamp(self):
|
||||
""" Epoch milliseconds
|
||||
"""
|
||||
return self._timestamp
|
||||
|
||||
@property
|
||||
def timestamp_type(self):
|
||||
""" CREATE_TIME(0) or APPEND_TIME(1)
|
||||
"""
|
||||
return self._timestamp_type
|
||||
|
||||
@property
|
||||
def key(self):
|
||||
""" Bytes key or None
|
||||
"""
|
||||
return self._key
|
||||
|
||||
@property
|
||||
def value(self):
|
||||
""" Bytes value or None
|
||||
"""
|
||||
return self._value
|
||||
|
||||
@property
|
||||
def headers(self):
|
||||
return self._headers
|
||||
|
||||
@property
|
||||
def checksum(self):
|
||||
return None
|
||||
|
||||
def validate_crc(self):
|
||||
return True
|
||||
|
||||
def __repr__(self):
|
||||
return (
|
||||
"DefaultRecord(offset={!r}, timestamp={!r}, timestamp_type={!r},"
|
||||
" key={!r}, value={!r}, headers={!r})".format(
|
||||
self._offset, self._timestamp, self._timestamp_type,
|
||||
self._key, self._value, self._headers)
|
||||
)
|
||||
|
||||
|
||||
class ControlRecord(DefaultRecord):
|
||||
__slots__ = ("_size_in_bytes", "_offset", "_timestamp", "_timestamp_type", "_key", "_value",
|
||||
"_headers", "_version", "_type")
|
||||
|
||||
KEY_STRUCT = struct.Struct(
|
||||
">h" # Current Version => Int16
|
||||
"h" # Type => Int16 (0 indicates an abort marker, 1 indicates a commit)
|
||||
)
|
||||
|
||||
def __init__(self, size_in_bytes, offset, timestamp, timestamp_type, key, value, headers):
|
||||
super(ControlRecord, self).__init__(size_in_bytes, offset, timestamp, timestamp_type, key, value, headers)
|
||||
(self._version, self._type) = self.KEY_STRUCT.unpack(self._key)
|
||||
|
||||
# see https://kafka.apache.org/documentation/#controlbatch
|
||||
@property
|
||||
def version(self):
|
||||
return self._version
|
||||
|
||||
@property
|
||||
def type(self):
|
||||
return self._type
|
||||
|
||||
@property
|
||||
def abort(self):
|
||||
return self._type == 0
|
||||
|
||||
@property
|
||||
def commit(self):
|
||||
return self._type == 1
|
||||
|
||||
def __repr__(self):
|
||||
return (
|
||||
"ControlRecord(offset={!r}, timestamp={!r}, timestamp_type={!r},"
|
||||
" version={!r}, type={!r} <{!s}>)".format(
|
||||
self._offset, self._timestamp, self._timestamp_type,
|
||||
self._version, self._type, "abort" if self.abort else "commit")
|
||||
)
|
||||
|
||||
|
||||
class DefaultRecordBatchBuilder(DefaultRecordBase, ABCRecordBatchBuilder):
|
||||
|
||||
# excluding key, value and headers:
|
||||
# 5 bytes length + 10 bytes timestamp + 5 bytes offset + 1 byte attributes
|
||||
MAX_RECORD_OVERHEAD = 21
|
||||
|
||||
__slots__ = ("_magic", "_compression_type", "_batch_size", "_is_transactional",
|
||||
"_producer_id", "_producer_epoch", "_base_sequence",
|
||||
"_first_timestamp", "_max_timestamp", "_last_offset", "_num_records",
|
||||
"_buffer")
|
||||
|
||||
def __init__(
|
||||
self, magic, compression_type, is_transactional,
|
||||
producer_id, producer_epoch, base_sequence, batch_size):
|
||||
assert magic >= 2
|
||||
self._magic = magic
|
||||
self._compression_type = compression_type & self.CODEC_MASK
|
||||
self._batch_size = batch_size
|
||||
self._is_transactional = bool(is_transactional)
|
||||
# KIP-98 fields for EOS
|
||||
self._producer_id = producer_id
|
||||
self._producer_epoch = producer_epoch
|
||||
self._base_sequence = base_sequence
|
||||
|
||||
self._first_timestamp = None
|
||||
self._max_timestamp = None
|
||||
self._last_offset = 0
|
||||
self._num_records = 0
|
||||
|
||||
self._buffer = bytearray(self.HEADER_STRUCT.size)
|
||||
|
||||
def set_producer_state(self, producer_id, producer_epoch, base_sequence, is_transactional):
|
||||
assert not is_transactional or producer_id != -1, "Cannot write transactional messages without a valid producer ID"
|
||||
assert producer_id == -1 or producer_epoch != -1, "Invalid negative producer epoch"
|
||||
assert producer_id == -1 or base_sequence != -1, "Invalid negative sequence number"
|
||||
self._producer_id = producer_id
|
||||
self._producer_epoch = producer_epoch
|
||||
self._base_sequence = base_sequence
|
||||
self._is_transactional = is_transactional
|
||||
|
||||
@property
|
||||
def producer_id(self):
|
||||
return self._producer_id
|
||||
|
||||
@property
|
||||
def producer_epoch(self):
|
||||
return self._producer_epoch
|
||||
|
||||
def _get_attributes(self, include_compression_type=True):
|
||||
attrs = 0
|
||||
if include_compression_type:
|
||||
attrs |= self._compression_type
|
||||
# Timestamp Type is set by Broker
|
||||
if self._is_transactional:
|
||||
attrs |= self.TRANSACTIONAL_MASK
|
||||
# Control batches are only created by Broker
|
||||
return attrs
|
||||
|
||||
def append(self, offset, timestamp, key, value, headers,
|
||||
# Cache for LOAD_FAST opcodes
|
||||
encode_varint=encode_varint, size_of_varint=size_of_varint,
|
||||
get_type=type, type_int=int, time_time=time.time,
|
||||
byte_like=(bytes, bytearray, memoryview),
|
||||
bytearray_type=bytearray, len_func=len, zero_len_varint=1
|
||||
):
|
||||
""" Write message to messageset buffer with MsgVersion 2
|
||||
"""
|
||||
# Check types
|
||||
if get_type(offset) != type_int:
|
||||
raise TypeError(offset)
|
||||
if timestamp is None:
|
||||
timestamp = type_int(time_time() * 1000)
|
||||
elif get_type(timestamp) != type_int:
|
||||
raise TypeError(timestamp)
|
||||
if not (key is None or get_type(key) in byte_like):
|
||||
raise TypeError(
|
||||
"Not supported type for key: {}".format(type(key)))
|
||||
if not (value is None or get_type(value) in byte_like):
|
||||
raise TypeError(
|
||||
"Not supported type for value: {}".format(type(value)))
|
||||
|
||||
# We will always add the first message, so those will be set
|
||||
if self._first_timestamp is None:
|
||||
self._first_timestamp = timestamp
|
||||
self._max_timestamp = timestamp
|
||||
timestamp_delta = 0
|
||||
first_message = 1
|
||||
else:
|
||||
timestamp_delta = timestamp - self._first_timestamp
|
||||
first_message = 0
|
||||
|
||||
# We can't write record right away to out buffer, we need to
|
||||
# precompute the length as first value...
|
||||
message_buffer = bytearray_type(b"\x00") # Attributes
|
||||
write_byte = message_buffer.append
|
||||
write = message_buffer.extend
|
||||
|
||||
encode_varint(timestamp_delta, write_byte)
|
||||
# Base offset is always 0 on Produce
|
||||
encode_varint(offset, write_byte)
|
||||
|
||||
if key is not None:
|
||||
encode_varint(len_func(key), write_byte)
|
||||
write(key)
|
||||
else:
|
||||
write_byte(zero_len_varint)
|
||||
|
||||
if value is not None:
|
||||
encode_varint(len_func(value), write_byte)
|
||||
write(value)
|
||||
else:
|
||||
write_byte(zero_len_varint)
|
||||
|
||||
encode_varint(len_func(headers), write_byte)
|
||||
|
||||
for h_key, h_value in headers:
|
||||
h_key = h_key.encode("utf-8")
|
||||
encode_varint(len_func(h_key), write_byte)
|
||||
write(h_key)
|
||||
if h_value is not None:
|
||||
encode_varint(len_func(h_value), write_byte)
|
||||
write(h_value)
|
||||
else:
|
||||
write_byte(zero_len_varint)
|
||||
|
||||
message_len = len_func(message_buffer)
|
||||
main_buffer = self._buffer
|
||||
|
||||
required_size = message_len + size_of_varint(message_len)
|
||||
# Check if we can write this message
|
||||
if (required_size + len_func(main_buffer) > self._batch_size and
|
||||
not first_message):
|
||||
return None
|
||||
|
||||
# Those should be updated after the length check
|
||||
if self._max_timestamp < timestamp:
|
||||
self._max_timestamp = timestamp
|
||||
self._num_records += 1
|
||||
self._last_offset = offset
|
||||
|
||||
encode_varint(message_len, main_buffer.append)
|
||||
main_buffer.extend(message_buffer)
|
||||
|
||||
return DefaultRecordMetadata(offset, required_size, timestamp)
|
||||
|
||||
def write_header(self, use_compression_type=True):
|
||||
batch_len = len(self._buffer)
|
||||
self.HEADER_STRUCT.pack_into(
|
||||
self._buffer, 0,
|
||||
0, # BaseOffset, set by broker
|
||||
batch_len - self.AFTER_LEN_OFFSET, # Size from here to end
|
||||
0, # PartitionLeaderEpoch, set by broker
|
||||
self._magic,
|
||||
0, # CRC will be set below, as we need a filled buffer for it
|
||||
self._get_attributes(use_compression_type),
|
||||
self._last_offset,
|
||||
self._first_timestamp or 0,
|
||||
self._max_timestamp or 0,
|
||||
self._producer_id,
|
||||
self._producer_epoch,
|
||||
self._base_sequence,
|
||||
self._num_records
|
||||
)
|
||||
crc = calc_crc32c(self._buffer[self.ATTRIBUTES_OFFSET:])
|
||||
struct.pack_into(">I", self._buffer, self.CRC_OFFSET, crc)
|
||||
|
||||
def _maybe_compress(self):
|
||||
if self._compression_type != self.CODEC_NONE:
|
||||
self._assert_has_codec(self._compression_type)
|
||||
header_size = self.HEADER_STRUCT.size
|
||||
data = bytes(self._buffer[header_size:])
|
||||
if self._compression_type == self.CODEC_GZIP:
|
||||
compressed = gzip_encode(data)
|
||||
elif self._compression_type == self.CODEC_SNAPPY:
|
||||
compressed = snappy_encode(data)
|
||||
elif self._compression_type == self.CODEC_LZ4:
|
||||
compressed = lz4_encode(data)
|
||||
elif self._compression_type == self.CODEC_ZSTD:
|
||||
compressed = zstd_encode(data)
|
||||
compressed_size = len(compressed)
|
||||
if len(data) <= compressed_size:
|
||||
# We did not get any benefit from compression, lets send
|
||||
# uncompressed
|
||||
return False
|
||||
else:
|
||||
# Trim bytearray to the required size
|
||||
needed_size = header_size + compressed_size
|
||||
del self._buffer[needed_size:]
|
||||
self._buffer[header_size:needed_size] = compressed
|
||||
return True
|
||||
return False
|
||||
|
||||
def build(self):
|
||||
send_compressed = self._maybe_compress()
|
||||
self.write_header(send_compressed)
|
||||
return self._buffer
|
||||
|
||||
def size(self):
|
||||
""" Return current size of data written to buffer
|
||||
"""
|
||||
return len(self._buffer)
|
||||
|
||||
@classmethod
|
||||
def header_size_in_bytes(self):
|
||||
return self.HEADER_STRUCT.size
|
||||
|
||||
@classmethod
|
||||
def size_in_bytes(self, offset_delta, timestamp_delta, key, value, headers):
|
||||
size_of_body = (
|
||||
1 + # Attrs
|
||||
size_of_varint(offset_delta) +
|
||||
size_of_varint(timestamp_delta) +
|
||||
self.size_of(key, value, headers)
|
||||
)
|
||||
return size_of_body + size_of_varint(size_of_body)
|
||||
|
||||
@classmethod
|
||||
def size_of(cls, key, value, headers):
|
||||
size = 0
|
||||
# Key size
|
||||
if key is None:
|
||||
size += 1
|
||||
else:
|
||||
key_len = len(key)
|
||||
size += size_of_varint(key_len) + key_len
|
||||
# Value size
|
||||
if value is None:
|
||||
size += 1
|
||||
else:
|
||||
value_len = len(value)
|
||||
size += size_of_varint(value_len) + value_len
|
||||
# Header size
|
||||
size += size_of_varint(len(headers))
|
||||
for h_key, h_value in headers:
|
||||
h_key_len = len(h_key.encode("utf-8"))
|
||||
size += size_of_varint(h_key_len) + h_key_len
|
||||
|
||||
if h_value is None:
|
||||
size += 1
|
||||
else:
|
||||
h_value_len = len(h_value)
|
||||
size += size_of_varint(h_value_len) + h_value_len
|
||||
return size
|
||||
|
||||
@classmethod
|
||||
def estimate_size_in_bytes(cls, key, value, headers):
|
||||
""" Get the upper bound estimate on the size of record
|
||||
"""
|
||||
return (
|
||||
cls.HEADER_STRUCT.size + cls.MAX_RECORD_OVERHEAD +
|
||||
cls.size_of(key, value, headers)
|
||||
)
|
||||
|
||||
def __str__(self):
|
||||
return (
|
||||
"DefaultRecordBatchBuilder(magic={}, base_offset={}, last_offset_delta={},"
|
||||
" first_timestamp={}, max_timestamp={},"
|
||||
" is_transactional={}, producer_id={}, producer_epoch={}, base_sequence={},"
|
||||
" records_count={})".format(
|
||||
self._magic, 0, self._last_offset,
|
||||
self._first_timestamp or 0, self._max_timestamp or 0,
|
||||
self._is_transactional, self._producer_id, self._producer_epoch, self._base_sequence,
|
||||
self._num_records))
|
||||
|
||||
|
||||
class DefaultRecordMetadata(object):
|
||||
|
||||
__slots__ = ("_size", "_timestamp", "_offset")
|
||||
|
||||
def __init__(self, offset, size, timestamp):
|
||||
self._offset = offset
|
||||
self._size = size
|
||||
self._timestamp = timestamp
|
||||
|
||||
@property
|
||||
def offset(self):
|
||||
return self._offset
|
||||
|
||||
@property
|
||||
def crc(self):
|
||||
return None
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
return self._size
|
||||
|
||||
@property
|
||||
def timestamp(self):
|
||||
return self._timestamp
|
||||
|
||||
def __repr__(self):
|
||||
return (
|
||||
"DefaultRecordMetadata(offset={!r}, size={!r}, timestamp={!r})"
|
||||
.format(self._offset, self._size, self._timestamp)
|
||||
)
|
||||
@@ -0,0 +1,580 @@
|
||||
# See:
|
||||
# https://github.com/apache/kafka/blob/trunk/clients/src/main/java/org/\
|
||||
# apache/kafka/common/record/LegacyRecord.java
|
||||
|
||||
# Builder and reader implementation for V0 and V1 record versions. As of Kafka
|
||||
# 0.11.0.0 those were replaced with V2, thus the Legacy naming.
|
||||
|
||||
# The schema is given below (see
|
||||
# https://kafka.apache.org/protocol#protocol_message_sets for more details):
|
||||
|
||||
# MessageSet => [Offset MessageSize Message]
|
||||
# Offset => int64
|
||||
# MessageSize => int32
|
||||
|
||||
# v0
|
||||
# Message => Crc MagicByte Attributes Key Value
|
||||
# Crc => int32
|
||||
# MagicByte => int8
|
||||
# Attributes => int8
|
||||
# Key => bytes
|
||||
# Value => bytes
|
||||
|
||||
# v1 (supported since 0.10.0)
|
||||
# Message => Crc MagicByte Attributes Key Value
|
||||
# Crc => int32
|
||||
# MagicByte => int8
|
||||
# Attributes => int8
|
||||
# Timestamp => int64
|
||||
# Key => bytes
|
||||
# Value => bytes
|
||||
|
||||
# The message attribute bits are given below:
|
||||
# * Unused (4-7)
|
||||
# * Timestamp Type (3) (added in V1)
|
||||
# * Compression Type (0-2)
|
||||
|
||||
# Note that when compression is enabled (see attributes above), the whole
|
||||
# array of MessageSet's is compressed and places into a message as the `value`.
|
||||
# Only the parent message is marked with `compression` bits in attributes.
|
||||
|
||||
# The CRC covers the data from the Magic byte to the end of the message.
|
||||
|
||||
|
||||
import struct
|
||||
import time
|
||||
|
||||
from kafka.record.abc import ABCRecord, ABCRecordBatch, ABCRecordBatchBuilder
|
||||
from kafka.record.util import calc_crc32
|
||||
|
||||
from kafka.codec import (
|
||||
gzip_encode, snappy_encode, lz4_encode, lz4_encode_old_kafka,
|
||||
gzip_decode, snappy_decode, lz4_decode, lz4_decode_old_kafka,
|
||||
)
|
||||
import kafka.codec as codecs
|
||||
from kafka.errors import CorruptRecordError, UnsupportedCodecError
|
||||
|
||||
|
||||
class LegacyRecordBase(object):
|
||||
|
||||
__slots__ = ()
|
||||
|
||||
HEADER_STRUCT_V0 = struct.Struct(
|
||||
">q" # BaseOffset => Int64
|
||||
"i" # Length => Int32
|
||||
"I" # CRC => Int32
|
||||
"b" # Magic => Int8
|
||||
"b" # Attributes => Int8
|
||||
)
|
||||
HEADER_STRUCT_V1 = struct.Struct(
|
||||
">q" # BaseOffset => Int64
|
||||
"i" # Length => Int32
|
||||
"I" # CRC => Int32
|
||||
"b" # Magic => Int8
|
||||
"b" # Attributes => Int8
|
||||
"q" # timestamp => Int64
|
||||
)
|
||||
|
||||
LOG_OVERHEAD = CRC_OFFSET = struct.calcsize(
|
||||
">q" # Offset
|
||||
"i" # Size
|
||||
)
|
||||
MAGIC_OFFSET = LOG_OVERHEAD + struct.calcsize(
|
||||
">I" # CRC
|
||||
)
|
||||
# Those are used for fast size calculations
|
||||
RECORD_OVERHEAD_V0 = struct.calcsize(
|
||||
">I" # CRC
|
||||
"b" # magic
|
||||
"b" # attributes
|
||||
"i" # Key length
|
||||
"i" # Value length
|
||||
)
|
||||
RECORD_OVERHEAD_V1 = struct.calcsize(
|
||||
">I" # CRC
|
||||
"b" # magic
|
||||
"b" # attributes
|
||||
"q" # timestamp
|
||||
"i" # Key length
|
||||
"i" # Value length
|
||||
)
|
||||
|
||||
KEY_OFFSET_V0 = HEADER_STRUCT_V0.size
|
||||
KEY_OFFSET_V1 = HEADER_STRUCT_V1.size
|
||||
KEY_LENGTH = VALUE_LENGTH = struct.calcsize(">i") # Bytes length is Int32
|
||||
|
||||
CODEC_MASK = 0x07
|
||||
CODEC_NONE = 0x00
|
||||
CODEC_GZIP = 0x01
|
||||
CODEC_SNAPPY = 0x02
|
||||
CODEC_LZ4 = 0x03
|
||||
TIMESTAMP_TYPE_MASK = 0x08
|
||||
|
||||
LOG_APPEND_TIME = 1
|
||||
CREATE_TIME = 0
|
||||
|
||||
NO_TIMESTAMP = -1
|
||||
|
||||
def _assert_has_codec(self, compression_type):
|
||||
if compression_type == self.CODEC_GZIP:
|
||||
checker, name = codecs.has_gzip, "gzip"
|
||||
elif compression_type == self.CODEC_SNAPPY:
|
||||
checker, name = codecs.has_snappy, "snappy"
|
||||
elif compression_type == self.CODEC_LZ4:
|
||||
checker, name = codecs.has_lz4, "lz4"
|
||||
if not checker():
|
||||
raise UnsupportedCodecError(
|
||||
"Libraries for {} compression codec not found".format(name))
|
||||
|
||||
|
||||
class LegacyRecordBatch(ABCRecordBatch, LegacyRecordBase):
|
||||
|
||||
__slots__ = ("_buffer", "_magic", "_offset", "_length", "_crc", "_timestamp",
|
||||
"_attributes", "_decompressed")
|
||||
|
||||
def __init__(self, buffer, magic):
|
||||
self._buffer = memoryview(buffer)
|
||||
self._magic = magic
|
||||
|
||||
offset, length, crc, magic_, attrs, timestamp = self._read_header(0)
|
||||
assert length == len(buffer) - self.LOG_OVERHEAD
|
||||
assert magic == magic_
|
||||
|
||||
self._offset = offset
|
||||
self._length = length
|
||||
self._crc = crc
|
||||
self._timestamp = timestamp
|
||||
self._attributes = attrs
|
||||
self._decompressed = False
|
||||
|
||||
@property
|
||||
def base_offset(self):
|
||||
return self._offset
|
||||
|
||||
@property
|
||||
def size_in_bytes(self):
|
||||
return self._length + self.LOG_OVERHEAD
|
||||
|
||||
@property
|
||||
def timestamp_type(self):
|
||||
"""0 for CreateTime; 1 for LogAppendTime; None if unsupported.
|
||||
|
||||
Value is determined by broker; produced messages should always set to 0
|
||||
Requires Kafka >= 0.10 / message version >= 1
|
||||
"""
|
||||
if self._magic == 0:
|
||||
return None
|
||||
elif self._attributes & self.TIMESTAMP_TYPE_MASK:
|
||||
return 1
|
||||
else:
|
||||
return 0
|
||||
|
||||
@property
|
||||
def compression_type(self):
|
||||
return self._attributes & self.CODEC_MASK
|
||||
|
||||
@property
|
||||
def magic(self):
|
||||
return self._magic
|
||||
|
||||
def validate_crc(self):
|
||||
crc = calc_crc32(self._buffer[self.MAGIC_OFFSET:])
|
||||
return self._crc == crc
|
||||
|
||||
def _decompress(self, key_offset):
|
||||
# Copy of `_read_key_value`, but uses memoryview
|
||||
pos = key_offset
|
||||
key_size = struct.unpack_from(">i", self._buffer, pos)[0]
|
||||
pos += self.KEY_LENGTH
|
||||
if key_size != -1:
|
||||
pos += key_size
|
||||
value_size = struct.unpack_from(">i", self._buffer, pos)[0]
|
||||
pos += self.VALUE_LENGTH
|
||||
if value_size == -1:
|
||||
raise CorruptRecordError("Value of compressed message is None")
|
||||
else:
|
||||
data = self._buffer[pos:pos + value_size]
|
||||
|
||||
compression_type = self.compression_type
|
||||
self._assert_has_codec(compression_type)
|
||||
if compression_type == self.CODEC_GZIP:
|
||||
uncompressed = gzip_decode(data)
|
||||
elif compression_type == self.CODEC_SNAPPY:
|
||||
uncompressed = snappy_decode(data.tobytes())
|
||||
elif compression_type == self.CODEC_LZ4:
|
||||
if self._magic == 0:
|
||||
uncompressed = lz4_decode_old_kafka(data.tobytes())
|
||||
else:
|
||||
uncompressed = lz4_decode(data.tobytes())
|
||||
return uncompressed
|
||||
|
||||
def _read_header(self, pos):
|
||||
if self._magic == 0:
|
||||
offset, length, crc, magic_read, attrs = \
|
||||
self.HEADER_STRUCT_V0.unpack_from(self._buffer, pos)
|
||||
timestamp = None
|
||||
else:
|
||||
offset, length, crc, magic_read, attrs, timestamp = \
|
||||
self.HEADER_STRUCT_V1.unpack_from(self._buffer, pos)
|
||||
return offset, length, crc, magic_read, attrs, timestamp
|
||||
|
||||
def _read_all_headers(self):
|
||||
pos = 0
|
||||
msgs = []
|
||||
buffer_len = len(self._buffer)
|
||||
while pos < buffer_len:
|
||||
header = self._read_header(pos)
|
||||
msgs.append((header, pos))
|
||||
pos += self.LOG_OVERHEAD + header[1] # length
|
||||
return msgs
|
||||
|
||||
def _read_key_value(self, pos):
|
||||
key_size = struct.unpack_from(">i", self._buffer, pos)[0]
|
||||
pos += self.KEY_LENGTH
|
||||
if key_size == -1:
|
||||
key = None
|
||||
else:
|
||||
key = self._buffer[pos:pos + key_size].tobytes()
|
||||
pos += key_size
|
||||
|
||||
value_size = struct.unpack_from(">i", self._buffer, pos)[0]
|
||||
pos += self.VALUE_LENGTH
|
||||
if value_size == -1:
|
||||
value = None
|
||||
else:
|
||||
value = self._buffer[pos:pos + value_size].tobytes()
|
||||
return key, value
|
||||
|
||||
def _crc_bytes(self, msg_pos, length):
|
||||
return self._buffer[msg_pos + self.MAGIC_OFFSET:msg_pos + self.LOG_OVERHEAD + length]
|
||||
|
||||
def __iter__(self):
|
||||
if self._magic == 1:
|
||||
key_offset = self.KEY_OFFSET_V1
|
||||
else:
|
||||
key_offset = self.KEY_OFFSET_V0
|
||||
timestamp_type = self.timestamp_type
|
||||
|
||||
if self.compression_type:
|
||||
# In case we will call iter again
|
||||
if not self._decompressed:
|
||||
self._buffer = memoryview(self._decompress(key_offset))
|
||||
self._decompressed = True
|
||||
|
||||
# If relative offset is used, we need to decompress the entire
|
||||
# message first to compute the absolute offset.
|
||||
headers = self._read_all_headers()
|
||||
if self._magic > 0:
|
||||
msg_header, _ = headers[-1]
|
||||
absolute_base_offset = self._offset - msg_header[0]
|
||||
else:
|
||||
absolute_base_offset = -1
|
||||
|
||||
for header, msg_pos in headers:
|
||||
offset, length, crc, _, attrs, timestamp = header
|
||||
# There should only ever be a single layer of compression
|
||||
assert not attrs & self.CODEC_MASK, (
|
||||
'MessageSet at offset %d appears double-compressed. This '
|
||||
'should not happen -- check your producers!' % (offset,))
|
||||
|
||||
# When magic value is greater than 0, the timestamp
|
||||
# of a compressed message depends on the
|
||||
# timestamp type of the wrapper message:
|
||||
if timestamp_type == self.LOG_APPEND_TIME:
|
||||
timestamp = self._timestamp
|
||||
|
||||
if absolute_base_offset >= 0:
|
||||
offset += absolute_base_offset
|
||||
|
||||
key, value = self._read_key_value(msg_pos + key_offset)
|
||||
crc_bytes = self._crc_bytes(msg_pos, length)
|
||||
yield LegacyRecord(
|
||||
self._magic, offset, timestamp, timestamp_type,
|
||||
key, value, crc, crc_bytes)
|
||||
else:
|
||||
key, value = self._read_key_value(key_offset)
|
||||
crc_bytes = self._crc_bytes(0, len(self._buffer) - self.LOG_OVERHEAD)
|
||||
yield LegacyRecord(
|
||||
self._magic, self._offset, self._timestamp, timestamp_type,
|
||||
key, value, self._crc, crc_bytes)
|
||||
|
||||
|
||||
class LegacyRecord(ABCRecord):
|
||||
|
||||
__slots__ = ("_magic", "_offset", "_timestamp", "_timestamp_type", "_key", "_value",
|
||||
"_crc", "_crc_bytes")
|
||||
|
||||
def __init__(self, magic, offset, timestamp, timestamp_type, key, value, crc, crc_bytes):
|
||||
self._magic = magic
|
||||
self._offset = offset
|
||||
self._timestamp = timestamp
|
||||
self._timestamp_type = timestamp_type
|
||||
self._key = key
|
||||
self._value = value
|
||||
self._crc = crc
|
||||
self._crc_bytes = crc_bytes
|
||||
|
||||
@property
|
||||
def magic(self):
|
||||
return self._magic
|
||||
|
||||
@property
|
||||
def offset(self):
|
||||
return self._offset
|
||||
|
||||
@property
|
||||
def timestamp(self):
|
||||
""" Epoch milliseconds
|
||||
"""
|
||||
return self._timestamp
|
||||
|
||||
@property
|
||||
def timestamp_type(self):
|
||||
""" CREATE_TIME(0) or APPEND_TIME(1)
|
||||
"""
|
||||
return self._timestamp_type
|
||||
|
||||
@property
|
||||
def key(self):
|
||||
""" Bytes key or None
|
||||
"""
|
||||
return self._key
|
||||
|
||||
@property
|
||||
def value(self):
|
||||
""" Bytes value or None
|
||||
"""
|
||||
return self._value
|
||||
|
||||
@property
|
||||
def headers(self):
|
||||
return []
|
||||
|
||||
@property
|
||||
def checksum(self):
|
||||
return self._crc
|
||||
|
||||
def validate_crc(self):
|
||||
crc = calc_crc32(self._crc_bytes)
|
||||
return self._crc == crc
|
||||
|
||||
@property
|
||||
def size_in_bytes(self):
|
||||
return LegacyRecordBatchBuilder.estimate_size_in_bytes(self._magic, None, self._key, self._value)
|
||||
|
||||
def __repr__(self):
|
||||
return (
|
||||
"LegacyRecord(magic={!r} offset={!r}, timestamp={!r}, timestamp_type={!r},"
|
||||
" key={!r}, value={!r}, crc={!r})".format(
|
||||
self._magic, self._offset, self._timestamp, self._timestamp_type,
|
||||
self._key, self._value, self._crc)
|
||||
)
|
||||
|
||||
|
||||
class LegacyRecordBatchBuilder(ABCRecordBatchBuilder, LegacyRecordBase):
|
||||
|
||||
__slots__ = ("_magic", "_compression_type", "_batch_size", "_buffer")
|
||||
|
||||
def __init__(self, magic, compression_type, batch_size):
|
||||
self._magic = magic
|
||||
self._compression_type = compression_type
|
||||
self._batch_size = batch_size
|
||||
self._buffer = bytearray()
|
||||
|
||||
def append(self, offset, timestamp, key, value, headers=None):
|
||||
""" Append message to batch.
|
||||
"""
|
||||
assert not headers, "Headers not supported in v0/v1"
|
||||
# Check types
|
||||
if type(offset) != int:
|
||||
raise TypeError(offset)
|
||||
if self._magic == 0:
|
||||
timestamp = self.NO_TIMESTAMP
|
||||
elif timestamp is None:
|
||||
timestamp = int(time.time() * 1000)
|
||||
elif type(timestamp) != int:
|
||||
raise TypeError(
|
||||
"`timestamp` should be int, but {} provided".format(
|
||||
type(timestamp)))
|
||||
if not (key is None or
|
||||
isinstance(key, (bytes, bytearray, memoryview))):
|
||||
raise TypeError(
|
||||
"Not supported type for key: {}".format(type(key)))
|
||||
if not (value is None or
|
||||
isinstance(value, (bytes, bytearray, memoryview))):
|
||||
raise TypeError(
|
||||
"Not supported type for value: {}".format(type(value)))
|
||||
|
||||
# Check if we have room for another message
|
||||
pos = len(self._buffer)
|
||||
size = self.size_in_bytes(offset, timestamp, key, value)
|
||||
# We always allow at least one record to be appended
|
||||
if offset != 0 and pos + size >= self._batch_size:
|
||||
return None
|
||||
|
||||
# Allocate proper buffer length
|
||||
self._buffer.extend(bytearray(size))
|
||||
|
||||
# Encode message
|
||||
crc = self._encode_msg(pos, offset, timestamp, key, value)
|
||||
|
||||
return LegacyRecordMetadata(offset, crc, size, timestamp)
|
||||
|
||||
def _encode_msg(self, start_pos, offset, timestamp, key, value,
|
||||
attributes=0):
|
||||
""" Encode msg data into the `msg_buffer`, which should be allocated
|
||||
to at least the size of this message.
|
||||
"""
|
||||
magic = self._magic
|
||||
buf = self._buffer
|
||||
pos = start_pos
|
||||
|
||||
# Write key and value
|
||||
pos += self.KEY_OFFSET_V0 if magic == 0 else self.KEY_OFFSET_V1
|
||||
|
||||
if key is None:
|
||||
struct.pack_into(">i", buf, pos, -1)
|
||||
pos += self.KEY_LENGTH
|
||||
else:
|
||||
key_size = len(key)
|
||||
struct.pack_into(">i", buf, pos, key_size)
|
||||
pos += self.KEY_LENGTH
|
||||
buf[pos: pos + key_size] = key
|
||||
pos += key_size
|
||||
|
||||
if value is None:
|
||||
struct.pack_into(">i", buf, pos, -1)
|
||||
pos += self.VALUE_LENGTH
|
||||
else:
|
||||
value_size = len(value)
|
||||
struct.pack_into(">i", buf, pos, value_size)
|
||||
pos += self.VALUE_LENGTH
|
||||
buf[pos: pos + value_size] = value
|
||||
pos += value_size
|
||||
length = (pos - start_pos) - self.LOG_OVERHEAD
|
||||
|
||||
# Write msg header. Note, that Crc will be updated later
|
||||
if magic == 0:
|
||||
self.HEADER_STRUCT_V0.pack_into(
|
||||
buf, start_pos,
|
||||
offset, length, 0, magic, attributes)
|
||||
else:
|
||||
self.HEADER_STRUCT_V1.pack_into(
|
||||
buf, start_pos,
|
||||
offset, length, 0, magic, attributes, timestamp)
|
||||
|
||||
# Calculate CRC for msg
|
||||
crc_data = memoryview(buf)[start_pos + self.MAGIC_OFFSET:]
|
||||
crc = calc_crc32(crc_data)
|
||||
struct.pack_into(">I", buf, start_pos + self.CRC_OFFSET, crc)
|
||||
return crc
|
||||
|
||||
def _maybe_compress(self):
|
||||
if self._compression_type:
|
||||
self._assert_has_codec(self._compression_type)
|
||||
data = bytes(self._buffer)
|
||||
if self._compression_type == self.CODEC_GZIP:
|
||||
compressed = gzip_encode(data)
|
||||
elif self._compression_type == self.CODEC_SNAPPY:
|
||||
compressed = snappy_encode(data)
|
||||
elif self._compression_type == self.CODEC_LZ4:
|
||||
if self._magic == 0:
|
||||
compressed = lz4_encode_old_kafka(data)
|
||||
else:
|
||||
compressed = lz4_encode(data)
|
||||
size = self.size_in_bytes(
|
||||
0, timestamp=0, key=None, value=compressed)
|
||||
# We will try to reuse the same buffer if we have enough space
|
||||
if size > len(self._buffer):
|
||||
self._buffer = bytearray(size)
|
||||
else:
|
||||
del self._buffer[size:]
|
||||
self._encode_msg(
|
||||
start_pos=0,
|
||||
offset=0, timestamp=0, key=None, value=compressed,
|
||||
attributes=self._compression_type)
|
||||
return True
|
||||
return False
|
||||
|
||||
def build(self):
|
||||
"""Compress batch to be ready for send"""
|
||||
self._maybe_compress()
|
||||
return self._buffer
|
||||
|
||||
def size(self):
|
||||
""" Return current size of data written to buffer
|
||||
"""
|
||||
return len(self._buffer)
|
||||
|
||||
# Size calculations. Just copied Java's implementation
|
||||
|
||||
def size_in_bytes(self, offset, timestamp, key, value, headers=None):
|
||||
""" Actual size of message to add
|
||||
"""
|
||||
assert not headers, "Headers not supported in v0/v1"
|
||||
magic = self._magic
|
||||
return self.LOG_OVERHEAD + self.record_size(magic, key, value)
|
||||
|
||||
@classmethod
|
||||
def record_size(cls, magic, key, value):
|
||||
message_size = cls.record_overhead(magic)
|
||||
if key is not None:
|
||||
message_size += len(key)
|
||||
if value is not None:
|
||||
message_size += len(value)
|
||||
return message_size
|
||||
|
||||
@classmethod
|
||||
def record_overhead(cls, magic):
|
||||
assert magic in [0, 1], "Not supported magic"
|
||||
if magic == 0:
|
||||
return cls.RECORD_OVERHEAD_V0
|
||||
else:
|
||||
return cls.RECORD_OVERHEAD_V1
|
||||
|
||||
@classmethod
|
||||
def estimate_size_in_bytes(cls, magic, compression_type, key, value):
|
||||
""" Upper bound estimate of record size.
|
||||
"""
|
||||
assert magic in [0, 1], "Not supported magic"
|
||||
# In case of compression we may need another overhead for inner msg
|
||||
if compression_type:
|
||||
return (
|
||||
cls.LOG_OVERHEAD + cls.record_overhead(magic) +
|
||||
cls.record_size(magic, key, value)
|
||||
)
|
||||
return cls.LOG_OVERHEAD + cls.record_size(magic, key, value)
|
||||
|
||||
|
||||
class LegacyRecordMetadata(object):
|
||||
|
||||
__slots__ = ("_crc", "_size", "_timestamp", "_offset")
|
||||
|
||||
def __init__(self, offset, crc, size, timestamp):
|
||||
self._offset = offset
|
||||
self._crc = crc
|
||||
self._size = size
|
||||
self._timestamp = timestamp
|
||||
|
||||
@property
|
||||
def offset(self):
|
||||
return self._offset
|
||||
|
||||
@property
|
||||
def crc(self):
|
||||
return self._crc
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
return self._size
|
||||
|
||||
@property
|
||||
def timestamp(self):
|
||||
return self._timestamp
|
||||
|
||||
def __repr__(self):
|
||||
return (
|
||||
"LegacyRecordMetadata(offset={!r}, crc={!r}, size={!r},"
|
||||
" timestamp={!r})".format(
|
||||
self._offset, self._crc, self._size, self._timestamp)
|
||||
)
|
||||
@@ -0,0 +1,239 @@
|
||||
# This class takes advantage of the fact that all formats v0, v1 and v2 of
|
||||
# messages storage has the same byte offsets for Length and Magic fields.
|
||||
# Lets look closely at what leading bytes all versions have:
|
||||
#
|
||||
# V0 and V1 (Offset is MessageSet part, other bytes are Message ones):
|
||||
# Offset => Int64
|
||||
# BytesLength => Int32
|
||||
# CRC => Int32
|
||||
# Magic => Int8
|
||||
# ...
|
||||
#
|
||||
# V2:
|
||||
# BaseOffset => Int64
|
||||
# Length => Int32
|
||||
# PartitionLeaderEpoch => Int32
|
||||
# Magic => Int8
|
||||
# ...
|
||||
#
|
||||
# So we can iterate over batches just by knowing offsets of Length. Magic is
|
||||
# used to construct the correct class for Batch itself.
|
||||
from __future__ import division
|
||||
|
||||
import struct
|
||||
|
||||
from kafka.errors import CorruptRecordError, IllegalStateError, UnsupportedVersionError
|
||||
from kafka.record.abc import ABCRecords
|
||||
from kafka.record.legacy_records import LegacyRecordBatch, LegacyRecordBatchBuilder
|
||||
from kafka.record.default_records import DefaultRecordBatch, DefaultRecordBatchBuilder
|
||||
|
||||
|
||||
class MemoryRecords(ABCRecords):
|
||||
|
||||
LENGTH_OFFSET = struct.calcsize(">q")
|
||||
LOG_OVERHEAD = struct.calcsize(">qi")
|
||||
MAGIC_OFFSET = struct.calcsize(">qii")
|
||||
|
||||
# Minimum space requirements for Record V0
|
||||
MIN_SLICE = LOG_OVERHEAD + LegacyRecordBatch.RECORD_OVERHEAD_V0
|
||||
|
||||
__slots__ = ("_buffer", "_pos", "_next_slice", "_remaining_bytes")
|
||||
|
||||
def __init__(self, bytes_data):
|
||||
self._buffer = bytes_data
|
||||
self._pos = 0
|
||||
# We keep one slice ahead so `has_next` will return very fast
|
||||
self._next_slice = None
|
||||
self._remaining_bytes = None
|
||||
self._cache_next()
|
||||
|
||||
def size_in_bytes(self):
|
||||
return len(self._buffer)
|
||||
|
||||
def valid_bytes(self):
|
||||
# We need to read the whole buffer to get the valid_bytes.
|
||||
# NOTE: in Fetcher we do the call after iteration, so should be fast
|
||||
if self._remaining_bytes is None:
|
||||
next_slice = self._next_slice
|
||||
pos = self._pos
|
||||
while self._remaining_bytes is None:
|
||||
self._cache_next()
|
||||
# Reset previous iterator position
|
||||
self._next_slice = next_slice
|
||||
self._pos = pos
|
||||
return len(self._buffer) - self._remaining_bytes
|
||||
|
||||
# NOTE: we cache offsets here as kwargs for a bit more speed, as cPython
|
||||
# will use LOAD_FAST opcode in this case
|
||||
def _cache_next(self, len_offset=LENGTH_OFFSET, log_overhead=LOG_OVERHEAD):
|
||||
buffer = self._buffer
|
||||
buffer_len = len(buffer)
|
||||
pos = self._pos
|
||||
remaining = buffer_len - pos
|
||||
if remaining < log_overhead:
|
||||
# Will be re-checked in Fetcher for remaining bytes.
|
||||
self._remaining_bytes = remaining
|
||||
self._next_slice = None
|
||||
return
|
||||
|
||||
length, = struct.unpack_from(
|
||||
">i", buffer, pos + len_offset)
|
||||
|
||||
slice_end = pos + log_overhead + length
|
||||
if slice_end > buffer_len:
|
||||
# Will be re-checked in Fetcher for remaining bytes
|
||||
self._remaining_bytes = remaining
|
||||
self._next_slice = None
|
||||
return
|
||||
|
||||
self._next_slice = memoryview(buffer)[pos: slice_end]
|
||||
self._pos = slice_end
|
||||
|
||||
def has_next(self):
|
||||
return self._next_slice is not None
|
||||
|
||||
# NOTE: same cache for LOAD_FAST as above
|
||||
def next_batch(self, _min_slice=MIN_SLICE,
|
||||
_magic_offset=MAGIC_OFFSET):
|
||||
next_slice = self._next_slice
|
||||
if next_slice is None:
|
||||
return None
|
||||
if len(next_slice) < _min_slice:
|
||||
raise CorruptRecordError(
|
||||
"Record size is less than the minimum record overhead "
|
||||
"({})".format(_min_slice - self.LOG_OVERHEAD))
|
||||
self._cache_next()
|
||||
magic, = struct.unpack_from(">b", next_slice, _magic_offset)
|
||||
if magic <= 1:
|
||||
return LegacyRecordBatch(next_slice, magic)
|
||||
else:
|
||||
return DefaultRecordBatch(next_slice)
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
||||
def __next__(self):
|
||||
if not self.has_next():
|
||||
raise StopIteration
|
||||
return self.next_batch()
|
||||
|
||||
next = __next__
|
||||
|
||||
|
||||
class MemoryRecordsBuilder(object):
|
||||
|
||||
__slots__ = ("_builder", "_batch_size", "_buffer", "_next_offset", "_closed",
|
||||
"_magic", "_bytes_written", "_producer_id", "_producer_epoch")
|
||||
|
||||
def __init__(self, magic, compression_type, batch_size, offset=0,
|
||||
transactional=False, producer_id=-1, producer_epoch=-1, base_sequence=-1):
|
||||
assert magic in [0, 1, 2], "Not supported magic"
|
||||
assert compression_type in [0, 1, 2, 3, 4], "Not valid compression type"
|
||||
if magic >= 2:
|
||||
assert not transactional or producer_id != -1, "Cannot write transactional messages without a valid producer ID"
|
||||
assert producer_id == -1 or producer_epoch != -1, "Invalid negative producer epoch"
|
||||
assert producer_id == -1 or base_sequence != -1, "Invalid negative sequence number used"
|
||||
|
||||
self._builder = DefaultRecordBatchBuilder(
|
||||
magic=magic, compression_type=compression_type,
|
||||
is_transactional=transactional, producer_id=producer_id,
|
||||
producer_epoch=producer_epoch, base_sequence=base_sequence,
|
||||
batch_size=batch_size)
|
||||
self._producer_id = producer_id
|
||||
self._producer_epoch = producer_epoch
|
||||
else:
|
||||
assert not transactional and producer_id == -1, "Idempotent messages are not supported for magic %s" % (magic,)
|
||||
self._builder = LegacyRecordBatchBuilder(
|
||||
magic=magic, compression_type=compression_type,
|
||||
batch_size=batch_size)
|
||||
self._producer_id = None
|
||||
self._batch_size = batch_size
|
||||
self._buffer = None
|
||||
|
||||
self._next_offset = offset
|
||||
self._closed = False
|
||||
self._magic = magic
|
||||
self._bytes_written = 0
|
||||
|
||||
def skip(self, offsets_to_skip):
|
||||
# Exposed for testing compacted records
|
||||
self._next_offset += offsets_to_skip
|
||||
|
||||
def append(self, timestamp, key, value, headers=[]):
|
||||
""" Append a message to the buffer.
|
||||
|
||||
Returns: RecordMetadata or None if unable to append
|
||||
"""
|
||||
if self._closed:
|
||||
return None
|
||||
|
||||
offset = self._next_offset
|
||||
metadata = self._builder.append(offset, timestamp, key, value, headers)
|
||||
# Return of None means there's no space to add a new message
|
||||
if metadata is None:
|
||||
return None
|
||||
|
||||
self._next_offset += 1
|
||||
return metadata
|
||||
|
||||
def set_producer_state(self, producer_id, producer_epoch, base_sequence, is_transactional):
|
||||
if self._magic < 2:
|
||||
raise UnsupportedVersionError('Producer State requires Message format v2+')
|
||||
elif self._closed:
|
||||
# Sequence numbers are assigned when the batch is closed while the accumulator is being drained.
|
||||
# If the resulting ProduceRequest to the partition leader failed for a retriable error, the batch will
|
||||
# be re queued. In this case, we should not attempt to set the state again, since changing the pid and sequence
|
||||
# once a batch has been sent to the broker risks introducing duplicates.
|
||||
raise IllegalStateError("Trying to set producer state of an already closed batch. This indicates a bug on the client.")
|
||||
self._builder.set_producer_state(producer_id, producer_epoch, base_sequence, is_transactional)
|
||||
self._producer_id = producer_id
|
||||
|
||||
@property
|
||||
def producer_id(self):
|
||||
return self._producer_id
|
||||
|
||||
@property
|
||||
def producer_epoch(self):
|
||||
return self._producer_epoch
|
||||
|
||||
def records(self):
|
||||
assert self._closed
|
||||
return MemoryRecords(self._buffer)
|
||||
|
||||
def close(self):
|
||||
# This method may be called multiple times on the same batch
|
||||
# i.e., on retries
|
||||
# we need to make sure we only close it out once
|
||||
# otherwise compressed messages may be double-compressed
|
||||
# see Issue 718
|
||||
if not self._closed:
|
||||
self._bytes_written = self._builder.size()
|
||||
self._buffer = bytes(self._builder.build())
|
||||
if self._magic == 2:
|
||||
self._producer_id = self._builder.producer_id
|
||||
self._producer_epoch = self._builder.producer_epoch
|
||||
self._builder = None
|
||||
self._closed = True
|
||||
|
||||
def size_in_bytes(self):
|
||||
if not self._closed:
|
||||
return self._builder.size()
|
||||
else:
|
||||
return len(self._buffer)
|
||||
|
||||
def compression_rate(self):
|
||||
assert self._closed
|
||||
return self.size_in_bytes() / self._bytes_written
|
||||
|
||||
def is_full(self):
|
||||
if self._closed:
|
||||
return True
|
||||
else:
|
||||
return self._builder.size() >= self._batch_size
|
||||
|
||||
def next_offset(self):
|
||||
return self._next_offset
|
||||
|
||||
def buffer(self):
|
||||
assert self._closed
|
||||
return self._buffer
|
||||
@@ -0,0 +1,135 @@
|
||||
import binascii
|
||||
|
||||
from kafka.record._crc32c import crc as crc32c_py
|
||||
try:
|
||||
from crc32c import crc32c as crc32c_c
|
||||
except ImportError:
|
||||
crc32c_c = None
|
||||
|
||||
|
||||
def encode_varint(value, write):
|
||||
""" Encode an integer to a varint presentation. See
|
||||
https://developers.google.com/protocol-buffers/docs/encoding?csw=1#varints
|
||||
on how those can be produced.
|
||||
|
||||
Arguments:
|
||||
value (int): Value to encode
|
||||
write (function): Called per byte that needs to be writen
|
||||
|
||||
Returns:
|
||||
int: Number of bytes written
|
||||
"""
|
||||
value = (value << 1) ^ (value >> 63)
|
||||
|
||||
if value <= 0x7f: # 1 byte
|
||||
write(value)
|
||||
return 1
|
||||
if value <= 0x3fff: # 2 bytes
|
||||
write(0x80 | (value & 0x7f))
|
||||
write(value >> 7)
|
||||
return 2
|
||||
if value <= 0x1fffff: # 3 bytes
|
||||
write(0x80 | (value & 0x7f))
|
||||
write(0x80 | ((value >> 7) & 0x7f))
|
||||
write(value >> 14)
|
||||
return 3
|
||||
if value <= 0xfffffff: # 4 bytes
|
||||
write(0x80 | (value & 0x7f))
|
||||
write(0x80 | ((value >> 7) & 0x7f))
|
||||
write(0x80 | ((value >> 14) & 0x7f))
|
||||
write(value >> 21)
|
||||
return 4
|
||||
if value <= 0x7ffffffff: # 5 bytes
|
||||
write(0x80 | (value & 0x7f))
|
||||
write(0x80 | ((value >> 7) & 0x7f))
|
||||
write(0x80 | ((value >> 14) & 0x7f))
|
||||
write(0x80 | ((value >> 21) & 0x7f))
|
||||
write(value >> 28)
|
||||
return 5
|
||||
else:
|
||||
# Return to general algorithm
|
||||
bits = value & 0x7f
|
||||
value >>= 7
|
||||
i = 0
|
||||
while value:
|
||||
write(0x80 | bits)
|
||||
bits = value & 0x7f
|
||||
value >>= 7
|
||||
i += 1
|
||||
write(bits)
|
||||
return i
|
||||
|
||||
|
||||
def size_of_varint(value):
|
||||
""" Number of bytes needed to encode an integer in variable-length format.
|
||||
"""
|
||||
value = (value << 1) ^ (value >> 63)
|
||||
if value <= 0x7f:
|
||||
return 1
|
||||
if value <= 0x3fff:
|
||||
return 2
|
||||
if value <= 0x1fffff:
|
||||
return 3
|
||||
if value <= 0xfffffff:
|
||||
return 4
|
||||
if value <= 0x7ffffffff:
|
||||
return 5
|
||||
if value <= 0x3ffffffffff:
|
||||
return 6
|
||||
if value <= 0x1ffffffffffff:
|
||||
return 7
|
||||
if value <= 0xffffffffffffff:
|
||||
return 8
|
||||
if value <= 0x7fffffffffffffff:
|
||||
return 9
|
||||
return 10
|
||||
|
||||
|
||||
def decode_varint(buffer, pos=0):
|
||||
""" Decode an integer from a varint presentation. See
|
||||
https://developers.google.com/protocol-buffers/docs/encoding?csw=1#varints
|
||||
on how those can be produced.
|
||||
|
||||
Arguments:
|
||||
buffer (bytearray): buffer to read from.
|
||||
pos (int): optional position to read from
|
||||
|
||||
Returns:
|
||||
(int, int): Decoded int value and next read position
|
||||
"""
|
||||
result = buffer[pos]
|
||||
if not (result & 0x81):
|
||||
return (result >> 1), pos + 1
|
||||
if not (result & 0x80):
|
||||
return (result >> 1) ^ (~0), pos + 1
|
||||
|
||||
result &= 0x7f
|
||||
pos += 1
|
||||
shift = 7
|
||||
while 1:
|
||||
b = buffer[pos]
|
||||
result |= ((b & 0x7f) << shift)
|
||||
pos += 1
|
||||
if not (b & 0x80):
|
||||
return ((result >> 1) ^ -(result & 1), pos)
|
||||
shift += 7
|
||||
if shift >= 64:
|
||||
raise ValueError("Out of int64 range")
|
||||
|
||||
|
||||
_crc32c = crc32c_py
|
||||
if crc32c_c is not None:
|
||||
_crc32c = crc32c_c
|
||||
|
||||
|
||||
def calc_crc32c(memview, _crc32c=_crc32c):
|
||||
""" Calculate CRC-32C (Castagnoli) checksum over a memoryview of data
|
||||
"""
|
||||
return _crc32c(memview)
|
||||
|
||||
|
||||
def calc_crc32(memview):
|
||||
""" Calculate simple CRC-32 checksum over a memoryview of data
|
||||
"""
|
||||
crc = binascii.crc32(memview) & 0xffffffff
|
||||
return crc
|
||||
Reference in New Issue
Block a user