This is an automated email from the ASF dual-hosted git repository. Cole-Greer pushed a commit to branch master in repository https://gitbox.apache.org/repos/asf/tinkerpop.git
commit 46edf5bca2c954c5d8f566e41257a6b933eebf04 Merge: 862bb93397 5fdfe38a66 Author: Cole Greer <[email protected]> AuthorDate: Tue Sep 29 15:06:18 2026 -0700 Merge branch '3.8-dev' .github/workflows/build-test.yml | 30 ++++++++--------- .github/workflows/codeql.yml | 14 ++++---- CHANGELOG.asciidoc | 2 ++ docs/src/reference/gremlin-variants.asciidoc | 5 +++ .../gremlin_python/structure/io/graphbinaryV4.py | 13 ++++++++ .../tests/unit/structure/io/test_graphbinaryV4.py | 39 ++++++++++++++++++++++ 6 files changed, 81 insertions(+), 22 deletions(-) diff --cc .github/workflows/build-test.yml index 889c1cc197,168ae63536..2c9138a49d --- a/.github/workflows/build-test.yml +++ b/.github/workflows/build-test.yml @@@ -20,10 -16,10 +20,10 @@@ jobs runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven run: mvn clean install -DskipTests -Dci --batch-mode -Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn @@@ -67,12 -62,32 +67,12 @@@ steps: - uses: actions/checkout@v7 - name: Set up JDK 25 -- uses: actions/setup-java@v5 - with: - java-version: '25' - distribution: 'temurin' - # spark-gremlin is excluded because no Spark release TinkerPop can adopt yet supports Java 25 (Spark 4.x, which - # adds Java 25 support, requires Java 17+ and Scala 2.13). hadoop-gremlin runs on Java 25 as of Hadoop 3.4.3. - - name: Build with Maven - run: mvn clean install -pl $EXCLUDE_MODULES,-:spark-gremlin -Dci --batch-mode -Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn - java-jdk11: - name: mvn clean install - jdk11 - timeout-minutes: 45 - needs: smoke - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - name: Set up JDK 11 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '25' distribution: 'temurin' - name: Build with Maven - run: mvn clean install -pl $EXCLUDE_MODULES -Dci --batch-mode -Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn -Dcoverage - - name: Upload to Codecov - uses: codecov/codecov-action@v7 - with: - directory: ./gremlin-tools/gremlin-coverage/target/site + run: mvn clean install -pl $EXCLUDE_MODULES -Dci --batch-mode -Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn gremlin-server-default: name: gremlin-server default timeout-minutes: 45 @@@ -80,10 -95,26 +80,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 - uses: actions/setup-java@v6 - with: - java-version: '11' - distribution: 'temurin' - - name: Build with Maven - run: | - mvn clean install -pl $EXCLUDE_MODULES -q -DskipTests -Dci - mvn verify -pl :gremlin-server -DskipTests -DskipIntegrationTests=false -DincludeNeo4j - gremlin-server-unified: - name: gremlin-server unified - timeout-minutes: 45 - needs: smoke - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven run: | @@@ -96,10 -127,10 +96,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Generate Gremlin Server Base working-directory: . @@@ -132,10 -163,10 +132,10 @@@ # runs-on: ubuntu-latest # steps: # - uses: actions/checkout@v7 -# - name: Set up JDK 11 +# - name: Set up JDK 17 - # uses: actions/setup-java@v5 + # uses: actions/setup-java@v6 # with: -# java-version: '11' +# java-version: '17' # distribution: 'temurin' # - name: Build with Maven # run: | @@@ -148,10 -179,10 +148,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven run: | @@@ -167,10 -198,10 +167,10 @@@ os: [ubuntu-latest, windows-latest] steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven Windows if: runner.os == 'Windows' @@@ -191,10 -222,10 +191,10 @@@ os: [ubuntu-latest, windows-latest] steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven Windows if: runner.os == 'Windows' @@@ -212,10 -243,10 +212,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven run: | @@@ -229,10 -260,10 +229,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Build with Maven run: | @@@ -246,13 -277,13 +246,13 @@@ strategy: fail-fast: false matrix: - node-version: [ '20', '22', '24', '26' ] + node-version: [ '22', '24', '26' ] steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Get Cached Server Base Image uses: actions/cache@v6 @@@ -283,10 -314,10 +283,10 @@@ PYTHON_VERSION: ${{ matrix.python-version }} steps: - uses: actions/checkout@v7 - - name: Set up JDK 11 + - name: Set up JDK 17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Set up Python ${{ matrix.python-version }} uses: actions/setup-python@v7 @@@ -305,10 -336,10 +305,10 @@@ runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - name: Set up JDK11 + - name: Set up JDK17 - uses: actions/setup-java@v5 + uses: actions/setup-java@v6 with: - java-version: '11' + java-version: '17' distribution: 'temurin' - name: Set up .NET 8.0.x uses: actions/setup-dotnet@v6 diff --cc gremlin-python/src/main/python/gremlin_python/structure/io/graphbinaryV4.py index 70b9ebf024,0000000000..5300919e64 mode 100644,000000..100644 --- a/gremlin-python/src/main/python/gremlin_python/structure/io/graphbinaryV4.py +++ b/gremlin-python/src/main/python/gremlin_python/structure/io/graphbinaryV4.py @@@ -1,1059 -1,0 +1,1072 @@@ +""" +Licensed to the Apache Software Foundation (ASF) under one +or more contributor license agreements. See the NOTICE file +distributed with this work for additional information +regarding copyright ownership. The ASF licenses this file +to you under the Apache License, Version 2.0 (the +"License"); you may not use this file except in compliance +with the License. You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, +software distributed under the License is distributed on an +"AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +KIND, either express or implied. See the License for the +specific language governing permissions and limitations +under the License. +""" +import calendar +import io +import logging +import math +import struct +import uuid +from collections import OrderedDict +from datetime import datetime, timedelta, timezone +from struct import pack, unpack + +from aenum import Enum +from gremlin_python.process.traversal import Direction, T, Merge +from gremlin_python.statics import FloatType, BigDecimal, ShortType, IntType, LongType, BigIntType, \ + DictType, SetType, SingleByte, SingleChar +from gremlin_python.structure.graph import Graph, Edge, Property, Vertex, VertexProperty, Path, CompositePDT, \ + PrimitivePDT, Tree, _pdt_decorated_types +from gremlin_python.structure.io.util import HashableDict, SymbolUtil, Marker + +log = logging.getLogger(__name__) + +# When we fall back to a superclass's serializer, we iterate over this map. +# We want that iteration order to be consistent, so we use an OrderedDict, +# not a dict. +_serializers = OrderedDict() +_deserializers = {} + + +class DataType(Enum): + null = 0xfe + int = 0x01 + long = 0x02 + string = 0x03 + datetime = 0x04 + double = 0x07 + float = 0x08 + list = 0x09 + map = 0x0a + set = 0x0b + uuid = 0x0c + edge = 0x0d + path = 0x0e + property = 0x0f + graph = 0x10 # not supported - no graph object in python yet + vertex = 0x11 + vertexproperty = 0x12 + direction = 0x18 + t = 0x20 + merge = 0x2e + bigdecimal = 0x22 + biginteger = 0x23 + byte = 0x24 + binary = 0x25 + short = 0x26 + boolean = 0x27 + tree = 0x2b + char = 0x80 + duration = 0x81 + composite_pdt = 0xf0 + primitive_pdt = 0xf1 + marker = 0xfd + + +NULL_BYTES = [DataType.null.value, 0x01] + +# null type code as a plain int, so the per-read null check skips the aenum lookup +_NULL = DataType.null.value + + +def _make_packer(format_string): + packer = struct.Struct(format_string) + pack = packer.pack + unpack = lambda s: packer.unpack(s)[0] + return pack, unpack + + +int64_pack, int64_unpack = _make_packer('>q') +int32_pack, int32_unpack = _make_packer('>i') +int16_pack, int16_unpack = _make_packer('>h') +int8_pack, int8_unpack = _make_packer('>b') +uint64_pack, uint64_unpack = _make_packer('>Q') +uint8_pack, uint8_unpack = _make_packer('>B') +float_pack, float_unpack = _make_packer('>f') +double_pack, double_unpack = _make_packer('>d') + + +class GraphBinaryTypeType(type): + def __new__(mcs, name, bases, dct): + cls = super(GraphBinaryTypeType, mcs).__new__(mcs, name, bases, dct) + if not name.startswith('_'): + if cls.python_type: + _serializers[cls.python_type] = cls + if cls.graphbinary_type: + _deserializers[cls.graphbinary_type] = cls + return cls + + +class GraphBinaryWriter(object): + def __init__(self, serializer_map=None): + self.serializers = _serializers.copy() + if serializer_map: + self.serializers.update(serializer_map) + + def write_object(self, object_data): + return self.to_dict(object_data) + + def to_dict(self, obj, to_extend=None): + if to_extend is None: + to_extend = bytearray() + + if obj is None: + to_extend.extend(NULL_BYTES) + return + + try: + t = type(obj) + return self.serializers[t].dictify(obj, self, to_extend) + except KeyError: + for key, serializer in self.serializers.items(): + if isinstance(obj, key): + return serializer.dictify(obj, self, to_extend) + + if isinstance(obj, dict): + return dict((self.to_dict(k, to_extend), self.to_dict(v, to_extend)) for k, v in obj.items()) + elif isinstance(obj, set): + return set([self.to_dict(o, to_extend) for o in obj]) + elif isinstance(obj, list): + return [self.to_dict(o, to_extend) for o in obj] + else: + return obj + + +class GraphBinaryReader(object): + def __init__(self, deserializer_map=None, pdt_registry=None): + self.deserializers = _deserializers.copy() + if deserializer_map: + self.deserializers.update(deserializer_map) + self.pdt_registry = pdt_registry + # Mirror of self.deserializers keyed by int type code instead of DataType. + # Avoids the per-read DataType(bt) call, whose aenum construction negatively affects performance on large results. + self._deserializer_by_type_code = {dt.value: des.objectify for dt, des in self.deserializers.items()} + + def read_object(self, b): + if b is None: + return None + if isinstance(b, bytearray): + return self.to_object(io.BytesIO(b)) + return self.to_object(b) + + def to_object(self, buff, data_type=None, nullable=True): + if data_type is None: + bt = uint8_unpack(buff.read(1)) + if bt == _NULL: + if nullable: + buff.read(1) + return None + try: + objectify = self._deserializer_by_type_code[bt] + except KeyError: + raise ValueError("%r is not a valid DataType" % bt) from None + result = objectify(buff, self, nullable) + else: + result = self.deserializers[data_type].objectify(buff, self, nullable) + if self.pdt_registry is not None and isinstance(result, PrimitivePDT): + hydrated = self.pdt_registry.hydrate_primitive(result) + if not isinstance(hydrated, PrimitivePDT): + return hydrated + result = hydrated + if self.pdt_registry is not None and isinstance(result, CompositePDT): + hydrated = self.pdt_registry.hydrate(result) + if not isinstance(hydrated, CompositePDT): + return hydrated + result = hydrated + if isinstance(result, CompositePDT) and result.name in _pdt_decorated_types: + return self._hydrate_decorated(result) + return result + + def _hydrate_decorated(self, pdt): + """Hydrate a CompositePDT using a @provider_defined decorated class.""" + cls = _pdt_decorated_types[pdt.name] + fields = {} + for k, v in pdt.fields.items(): + if isinstance(v, CompositePDT) and v.name in _pdt_decorated_types: + fields[k] = self._hydrate_decorated(v) + elif self.pdt_registry is not None and isinstance(v, CompositePDT): + fields[k] = self.pdt_registry.hydrate(v) + else: + fields[k] = v + obj = cls.__new__(cls) + for k, v in fields.items(): + setattr(obj, k, v) + return obj + + +class _GraphBinaryTypeIO(object, metaclass=GraphBinaryTypeType): + python_type = None + graphbinary_type = None + + @classmethod + def prefix_bytes(cls, graphbin_type, as_value=False, nullable=True, to_extend=None, ordered=False): + if to_extend is None: + to_extend = bytearray() + + if not as_value: + to_extend += uint8_pack(graphbin_type.value) + + if nullable: + if ordered: + to_extend += int8_pack(2) + else: + to_extend += int8_pack(0) + + return to_extend + + @classmethod + def read_int(cls, buff): + return int32_unpack(buff.read(4)) + + @classmethod + def is_null(cls, buff, reader, else_opt, nullable=True): + return None if nullable and buff.read(1)[0] == 0x01 else else_opt(buff, reader) + + def dictify(self, obj, writer, to_extend, as_value=False, nullable=True): + raise NotImplementedError() + + def objectify(self, d, reader, nullable=True): + raise NotImplementedError() + + +class LongIO(_GraphBinaryTypeIO): + + python_type = LongType + graphbinary_type = DataType.long + byte_format_pack = int64_pack + byte_format_unpack = int64_unpack + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + if obj < -9223372036854775808 or obj > 9223372036854775807: + raise Exception("Value too big, please use bigint Gremlin type") + else: + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(cls.byte_format_pack(obj)) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: int64_unpack(buff.read(8)), nullable) + + +class IntIO(LongIO): + + python_type = IntType + graphbinary_type = DataType.int + byte_format_pack = int32_pack + byte_format_unpack = int32_unpack + ++ @classmethod ++ def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): ++ # Python has one arbitrary precision int type, so every plain int arrives here ++ # regardless of magnitude. Anything outside the Java int range is written as a Long ++ # instead of reaching int32_pack, which fails with a bare struct.error. GraphSON ++ # coerces the same way (TINKERPOP-2360), and statics.long and statics.bigint stay ++ # available as explicit overrides. Promotion needs a type code in front of the ++ # value, so a value-only write keeps the fixed int32 layout. ++ if not as_value and not (-2147483648 <= obj <= 2147483647): ++ return LongIO.dictify(obj, writer, to_extend, as_value, nullable) ++ ++ return super().dictify(obj, writer, to_extend, as_value, nullable) ++ + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: cls.read_int(b), nullable) + + +class ShortIO(LongIO): + + python_type = ShortType + graphbinary_type = DataType.short + byte_format_pack = int16_pack + byte_format_unpack = int16_unpack + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: int16_unpack(buff.read(2)), nullable) + + +class BigIntIO(_GraphBinaryTypeIO): + + python_type = BigIntType + graphbinary_type = DataType.biginteger + + @classmethod + def write_bigint(cls, obj, to_extend): + # Compute the minimal signed two's-complement byte length, matching the + # Java reference serializer (BigInteger.toByteArray()). + bit_length = obj.bit_length() if obj >= 0 else (obj + 1).bit_length() + length = bit_length // 8 + 1 + b = obj.to_bytes(length, byteorder='big', signed=True) + to_extend.extend(int32_pack(length)) + to_extend.extend(b) + return to_extend + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + return cls.write_bigint(obj, to_extend) + + @classmethod + def read_bigint(cls, buff): + size = cls.read_int(buff) + return int.from_bytes(buff.read(size), byteorder='big', signed=True) + + @classmethod + def objectify(cls, buff, reader, nullable=False): + return cls.is_null(buff, reader, lambda b, r: cls.read_bigint(b), nullable) + + +def _long_bits_to_double(bits): + return unpack('d', pack('Q', bits))[0] + + +NAN = _long_bits_to_double(0x7ff8000000000000) +POSITIVE_INFINITY = _long_bits_to_double(0x7ff0000000000000) +NEGATIVE_INFINITY = _long_bits_to_double(0xFff0000000000000) + + +class FloatIO(LongIO): + + python_type = FloatType + graphbinary_type = DataType.float + graphbinary_base_type = DataType.float + byte_format_pack = float_pack + byte_format_unpack = float_unpack + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + if math.isnan(obj): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(cls.byte_format_pack(NAN)) + elif math.isinf(obj) and obj > 0: + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(cls.byte_format_pack(POSITIVE_INFINITY)) + elif math.isinf(obj) and obj < 0: + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(cls.byte_format_pack(NEGATIVE_INFINITY)) + else: + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(cls.byte_format_pack(obj)) + + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: float_unpack(b.read(4)), nullable) + + +class DoubleIO(FloatIO): + """ + Floats basically just fall through to double serialization. + """ + + graphbinary_type = DataType.double + graphbinary_base_type = DataType.double + byte_format_pack = double_pack + byte_format_unpack = double_unpack + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: double_unpack(b.read(8)), nullable) + + +class BigDecimalIO(_GraphBinaryTypeIO): + + python_type = BigDecimal + graphbinary_type = DataType.bigdecimal + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int32_pack(obj.scale)) + return BigIntIO.write_bigint(obj.unscaled_value, to_extend) + + @classmethod + def _read(cls, buff): + scale = int32_unpack(buff.read(4)) + unscaled_value = BigIntIO.read_bigint(buff) + return BigDecimal(scale, unscaled_value) + + @classmethod + def objectify(cls, buff, reader, nullable=False): + return cls.is_null(buff, reader, lambda b, r: cls._read(b), nullable) + + +class DateTimeIO(_GraphBinaryTypeIO): + + python_type = datetime + graphbinary_type = DataType.datetime + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + if obj.tzinfo is None: + raise AttributeError("Timezone information is required when constructing datetime") + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + IntIO.dictify(obj.year, writer, to_extend, True, False) + ByteIO.dictify(obj.month, writer, to_extend, True, False) + ByteIO.dictify(obj.day, writer, to_extend, True, False) + # construct time of day in nanoseconds + h = obj.time().hour + m = obj.time().minute + s = obj.time().second + ms = obj.time().microsecond + ns = round((h*60*60*1e9) + (m*60*1e9) + (s*1e9) + (ms*1e3)) + LongIO.dictify(ns, writer, to_extend, True, False) + os = round(obj.utcoffset().total_seconds()) + IntIO.dictify(os, writer, to_extend, True, False) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_dt, nullable) + + @classmethod + def _read_dt(cls, b, r): + year = r.to_object(b, DataType.int, False) + month = r.to_object(b, DataType.byte, False) + day = r.to_object(b, DataType.byte, False) + ns = r.to_object(b, DataType.long, False) + offset = r.to_object(b, DataType.int, False) + tz = timezone(timedelta(seconds=offset)) + return datetime(year, month, day, tzinfo=tz) + timedelta(microseconds=ns/1000) + + +class CharIO(_GraphBinaryTypeIO): + python_type = SingleChar + graphbinary_type = DataType.char + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(obj.encode("utf-8")) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_char, nullable) + + @classmethod + def _read_char(cls, b, r): + max_bytes = 4 + x = b.read(1) + while max_bytes > 0: + max_bytes = max_bytes - 1 + try: + return x.decode("utf-8") + except UnicodeDecodeError: + x += b.read(1) + + +class StringIO(_GraphBinaryTypeIO): + + python_type = str + graphbinary_type = DataType.string + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + str_bytes = obj.encode("utf-8") + to_extend += int32_pack(len(str_bytes)) + to_extend += str_bytes + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: b.read(cls.read_int(b)).decode("utf-8"), nullable) + + +class ListIO(_GraphBinaryTypeIO): + + python_type = list + graphbinary_type = DataType.list + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int32_pack(len(obj))) + for item in obj: + writer.to_dict(item, to_extend) + + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + flag = 0x00 + if nullable: + flag = buff.read(1)[0] + if flag == 0x01: + return None + else: + return cls._read_list(buff, reader, flag) + return cls._read_list(buff, reader, flag) + + @classmethod + def _read_list(cls, b, r, flag): + size = cls.read_int(b) + the_list = [] + if flag == 0x02: + while size > 0: + itm = r.read_object(b) + bulk = int64_unpack(b.read(8)) + for y in range(bulk): + the_list.append(itm) + size = size - 1 + else: + while size > 0: + the_list.append(r.read_object(b)) + size = size - 1 + + return the_list + + +class SetDeserializer(ListIO): + + python_type = SetType + graphbinary_type = DataType.set + + @classmethod + def objectify(cls, buff, reader, nullable=True): + the_list = ListIO.objectify(buff, reader, nullable) + try: + return set(the_list) + except TypeError: + log.warning("Coercing Set to list as it contains unhashable elements (e.g. dict, list). " + "See TINKERPOP-3232 for more details.") + return the_list + + +class MapIO(_GraphBinaryTypeIO): + + python_type = DictType + graphbinary_type = DataType.map + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend, ordered=isinstance(obj, OrderedDict)) + + to_extend.extend(int32_pack(len(obj))) + for k, v in obj.items(): + writer.to_dict(k, to_extend) + writer.to_dict(v, to_extend) + + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + flag = 0x00 + if nullable: + flag = buff.read(1)[0] + if flag == 0x01: + return None + else: + return cls._read_map(buff, reader, flag) + return cls._read_map(buff, reader, flag) + + @classmethod + def _read_map(cls, b, r, flag): + size = cls.read_int(b) + the_dict = OrderedDict() if flag == 0x02 else {} + while size > 0: + k = HashableDict.of(r.read_object(b)) + v = r.read_object(b) + the_dict[k] = v + size = size - 1 + + return the_dict + + +class UuidIO(_GraphBinaryTypeIO): + + python_type = uuid.UUID + graphbinary_type = DataType.uuid + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(obj.bytes) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: uuid.UUID(bytes=b.read(16)), nullable) + + +class EdgeIO(_GraphBinaryTypeIO): + + python_type = Edge + graphbinary_type = DataType.edge + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + + writer.to_dict(obj.id, to_extend) + # serializing labels as list according to GraphBinaryV4 + if hasattr(obj, '_labels'): + ListIO.dictify(list(obj._labels), writer, to_extend, True, False) + else: + ListIO.dictify([obj.label], writer, to_extend, True, False) + writer.to_dict(obj.inV.id, to_extend) + if hasattr(obj.inV, '_labels'): + ListIO.dictify(list(obj.inV._labels), writer, to_extend, True, False) + else: + ListIO.dictify([obj.inV.label], writer, to_extend, True, False) + writer.to_dict(obj.outV.id, to_extend) + if hasattr(obj.outV, '_labels'): + ListIO.dictify(list(obj.outV._labels), writer, to_extend, True, False) + else: + ListIO.dictify([obj.outV.label], writer, to_extend, True, False) + to_extend.extend(NULL_BYTES) + to_extend.extend(NULL_BYTES) + + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_edge, nullable) + + @classmethod + def _read_edge(cls, b, r): + edgeid = r.read_object(b) + # reading label list according to GraphBinaryV4 + edge_labels = r.to_object(b, DataType.list, False) + inv_id = r.read_object(b) + inv_labels = r.to_object(b, DataType.list, False) + inv = Vertex(inv_id, labels=inv_labels) + outv_id = r.read_object(b) + outv_labels = r.to_object(b, DataType.list, False) + outv = Vertex(outv_id, labels=outv_labels) + b.read(2) + props = r.read_object(b) + # null properties are returned as empty lists + properties = [] if props is None else props + edge = Edge(edgeid, outv, edge_labels[0] if edge_labels else "edge", inv, properties, labels=edge_labels) + return edge + + +class PathIO(_GraphBinaryTypeIO): + + python_type = Path + graphbinary_type = DataType.path + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + writer.to_dict(obj.labels, to_extend) + writer.to_dict(obj.objects, to_extend) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, lambda b, r: Path(r.read_object(b), r.read_object(b)), nullable) + + +class TreeIO(_GraphBinaryTypeIO): + + python_type = Tree + graphbinary_type = DataType.tree + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + # when as_value (a nested/bare child tree) prefix_bytes writes nothing: + # no type-id and no null flag. As a root value it writes {type-id}{null flag}. + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + root_nodes = obj.root_nodes() + to_extend.extend(int32_pack(len(root_nodes))) + for key in root_nodes: + child = obj.child_at(key) + # key is written fully-qualified (its own type-id + null flag + value) + writer.to_dict(key, to_extend) + # child is written as a BARE tree value: no type-id, no null flag + cls.dictify(child, writer, to_extend, as_value=True, nullable=False) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_tree, nullable) + + @classmethod + def _read_tree(cls, b, r): + size = cls.read_int(b) + tree = Tree() + while size > 0: + key = r.read_object(b) + child = cls.objectify(b, r, False) + tree.get_or_create_child(key).add_tree(child) + size = size - 1 + return tree + + +class PropertyIO(_GraphBinaryTypeIO): + + python_type = Property + graphbinary_type = DataType.property + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + StringIO.dictify(obj.key, writer, to_extend, True, False) + writer.to_dict(obj.value, to_extend) + to_extend.extend(NULL_BYTES) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_property, nullable) + + @classmethod + def _read_property(cls, b, r): + p = Property(r.to_object(b, DataType.string, False), r.read_object(b), None) + b.read(2) + return p + + +class TinkerGraphIO(_GraphBinaryTypeIO): + + python_type = Graph + graphbinary_type = DataType.graph + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + + vertices = list(obj.vertices.values()) + edges = list(obj.edges.values()) + + IntIO.dictify(len(vertices), writer, to_extend, True, False) + for v in vertices: + writer.to_dict(v.id, to_extend) + if hasattr(v, '_labels'): + ListIO.dictify(list(v._labels), writer, to_extend, True, False) + else: + ListIO.dictify([v.label], writer, to_extend, True, False) + v_props = v.properties + IntIO.dictify(len(v_props), writer, to_extend, True, False) + for vp in v_props: + writer.to_dict(vp.id, to_extend) + ListIO.dictify([vp.label], writer, to_extend, True, False) + writer.to_dict(vp.value, to_extend) + writer.to_dict(None, to_extend) + ListIO.dictify(vp.properties, writer, to_extend, True, False) + + IntIO.dictify(len(edges), writer, to_extend, True, False) + for e in edges: + writer.to_dict(e.id, to_extend) + if hasattr(e, '_labels'): + ListIO.dictify(list(e._labels), writer, to_extend, True, False) + else: + ListIO.dictify([e.label], writer, to_extend, True, False) + writer.to_dict(e.inV.id, to_extend) + writer.to_dict(None, to_extend) + writer.to_dict(e.outV.id, to_extend) + writer.to_dict(None, to_extend) + writer.to_dict(None, to_extend) + ListIO.dictify(e.properties, writer, to_extend, True, False) + + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_graph, nullable) + + @classmethod + def _read_graph(cls, b, r): + graph = Graph() + vertex_count = r.to_object(b, DataType.int, False) + for _ in range(vertex_count): + v_id = r.read_object(b) + v_labels = r.to_object(b, DataType.list, False) + vertex = Vertex(v_id, v_labels[0] if v_labels else "vertex", labels=v_labels) + graph.vertices[v_id] = vertex + + vp_count = r.to_object(b, DataType.int, False) + for _ in range(vp_count): + vp_id = r.read_object(b) + vp_label = r.to_object(b, DataType.list, False)[0] + vp_value = r.read_object(b) + r.read_object(b) # discard parent + vp = VertexProperty(vp_id, vp_label, vp_value, vertex) + vertex.properties.append(vp) + + meta_props = r.to_object(b, DataType.list, False) + if meta_props: + vp.properties.extend(meta_props) + + edge_count = r.to_object(b, DataType.int, False) + for _ in range(edge_count): + e_id = r.read_object(b) + e_labels = r.to_object(b, DataType.list, False) + in_v_id = r.read_object(b) + r.read_object(b) # discard in-v label + out_v_id = r.read_object(b) + r.read_object(b) # discard out-v label + r.read_object(b) # discard parent + + edge = Edge(e_id, graph.vertices[out_v_id], e_labels[0] if e_labels else "edge", + graph.vertices[in_v_id], labels=e_labels) + graph.edges[e_id] = edge + + edge_props = r.to_object(b, DataType.list, False) + if edge_props: + edge.properties.extend(edge_props) + + return graph + + +class VertexIO(_GraphBinaryTypeIO): + + python_type = Vertex + graphbinary_type = DataType.vertex + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + writer.to_dict(obj.id, to_extend) + # serializing labels as list according to GraphBinaryV4 + if hasattr(obj, '_labels'): + ListIO.dictify(list(obj._labels), writer, to_extend, True, False) + else: + ListIO.dictify([obj.label], writer, to_extend, True, False) + to_extend.extend(NULL_BYTES) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_vertex, nullable) + + @classmethod + def _read_vertex(cls, b, r): + vertex_id = r.read_object(b) + # reading label list according to GraphBinaryV4 + vertex_labels = r.to_object(b, DataType.list, False) + props = r.read_object(b) + # null properties are returned as empty lists + properties = [] if props is None else props + vertex = Vertex(vertex_id, properties=properties, labels=vertex_labels) + return vertex + + +class VertexPropertyIO(_GraphBinaryTypeIO): + + python_type = VertexProperty + graphbinary_type = DataType.vertexproperty + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + writer.to_dict(obj.id, to_extend) + # serializing label as list here for now according to GraphBinaryV4 + ListIO.dictify([obj.label], writer, to_extend, True, False) + writer.to_dict(obj.value, to_extend) + to_extend.extend(NULL_BYTES) + to_extend.extend(NULL_BYTES) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_vertexproperty, nullable) + + @classmethod + def _read_vertexproperty(cls, b, r): + # reading single string value for now according to GraphBinaryV4 + vp = VertexProperty(r.read_object(b), r.to_object(b, DataType.list, False)[0], r.read_object(b), None) + b.read(2) + properties = r.read_object(b) + # null properties are returned as empty lists + vp.properties = [] if properties is None else properties + return vp + + +class _EnumIO(_GraphBinaryTypeIO): + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + StringIO.dictify(SymbolUtil.to_camel_case(str(obj.name)), writer, to_extend) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_enumval, nullable) + + @classmethod + def _read_enumval(cls, b, r): + enum_name = r.to_object(b) + return cls.python_type[SymbolUtil.to_snake_case(enum_name)] + + +class DirectionIO(_EnumIO): + graphbinary_type = DataType.direction + python_type = Direction + + @classmethod + def _read_enumval(cls, b, r): + # Direction needs to retain all CAPS. note that to_/from_ are really just aliases of IN/OUT + # so they don't need to be accounted for in serialization + enum_name = r.to_object(b) + return cls.python_type[enum_name] + + +class TIO(_EnumIO): + graphbinary_type = DataType.t + python_type = T + + +class MergeIO(_EnumIO): + graphbinary_type = DataType.merge + python_type = Merge + + +class ByteIO(_GraphBinaryTypeIO): + python_type = SingleByte + graphbinary_type = DataType.byte + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int8_pack(obj)) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, + lambda b, r: int.__new__(SingleByte, int8_unpack(b.read(1))), + nullable) + + +class BinaryIO(_GraphBinaryTypeIO): + python_type = bytes + graphbinary_type = DataType.binary + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int32_pack(len(obj))) + to_extend.extend(obj) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_bytebuffer, nullable) + + @classmethod + def _read_bytebuffer(cls, b, r): + size = cls.read_int(b) + return bytes(b.read(size)) + + +class BooleanIO(_GraphBinaryTypeIO): + python_type = bool + graphbinary_type = DataType.boolean + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int8_pack(0x01 if obj else 0x00)) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, + lambda b, r: True if int8_unpack(b.read(1)) == 0x01 else False, + nullable) + + +class DurationIO(_GraphBinaryTypeIO): + python_type = timedelta + graphbinary_type = DataType.duration + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + LongIO.dictify(obj.seconds, writer, to_extend, True, False) + IntIO.dictify(obj.microseconds * 1000, writer, to_extend, True, False) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_duration, nullable) + + @classmethod + def _read_duration(cls, b, r): + seconds = r.to_object(b, DataType.long, False) + nanos = r.to_object(b, DataType.int, False) + return timedelta(seconds=seconds, microseconds=nanos / 1000) + + +class MarkerIO(_GraphBinaryTypeIO): + python_type = Marker + graphbinary_type = DataType.marker + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + to_extend.extend(int8_pack(obj.get_value())) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, + lambda b, r: Marker.of(int8_unpack(b.read(1))), + nullable) + + +class CompositePDTIO(_GraphBinaryTypeIO): + python_type = CompositePDT + graphbinary_type = DataType.composite_pdt + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + StringIO.dictify(obj.name, writer, to_extend) + MapIO.dictify(obj.fields, writer, to_extend) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_pdt, nullable) + + @classmethod + def _read_pdt(cls, b, r): + name = r.read_object(b) + fields = r.read_object(b) + return CompositePDT(name, fields) + + +class PrimitivePDTIO(_GraphBinaryTypeIO): + python_type = PrimitivePDT + graphbinary_type = DataType.primitive_pdt + + @classmethod + def dictify(cls, obj, writer, to_extend, as_value=False, nullable=True): + cls.prefix_bytes(cls.graphbinary_type, as_value, nullable, to_extend) + StringIO.dictify(obj.name, writer, to_extend) + StringIO.dictify(obj.value, writer, to_extend) + return to_extend + + @classmethod + def objectify(cls, buff, reader, nullable=True): + return cls.is_null(buff, reader, cls._read_primitive_pdt, nullable) + + @classmethod + def _read_primitive_pdt(cls, b, r): + name = r.read_object(b) + value = r.read_object(b) + return PrimitivePDT(name, value) diff --cc gremlin-python/src/main/python/tests/unit/structure/io/test_graphbinaryV4.py index 6208d0c321,0000000000..d6e9e1bf30 mode 100644,000000..100644 --- a/gremlin-python/src/main/python/tests/unit/structure/io/test_graphbinaryV4.py +++ b/gremlin-python/src/main/python/tests/unit/structure/io/test_graphbinaryV4.py @@@ -1,443 -1,0 +1,482 @@@ +""" +Licensed to the Apache Software Foundation (ASF) under one +or more contributor license agreements. See the NOTICE file +distributed with this work for additional information +regarding copyright ownership. The ASF licenses this file +to you under the Apache License, Version 2.0 (the +"License"); you may not use this file except in compliance +with the License. You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, +software distributed under the License is distributed on an +"AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +KIND, either express or implied. See the License for the +specific language governing permissions and limitations +under the License. +""" + +import uuid +import math ++import pytest +from collections import OrderedDict + +from datetime import datetime, timedelta, timezone +from gremlin_python.statics import long, bigint, BigDecimal, SingleByte, SingleChar +from gremlin_python.structure.graph import Graph, Vertex, Edge, Property, VertexProperty, Path, CompositePDT +from gremlin_python.structure.io.graphbinaryV4 import GraphBinaryWriter, GraphBinaryReader +from gremlin_python.process.traversal import Direction +from gremlin_python.structure.io.util import Marker + + +class TestGraphBinaryV4(object): + graphbinary_writer = GraphBinaryWriter() + graphbinary_reader = GraphBinaryReader() + + def test_null(self): + c = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(None)) + assert c is None + + def test_int(self): + x = 100 + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + ++ def test_int_out_of_int32_range_promotes_to_long(self): ++ # Python has one arbitrary precision int type, so a plain int beyond the Java int ++ # range used to reach int32_pack and fail with a bare struct.error (TINKERPOP-2363). ++ for x in [2 ** 31, -2 ** 31 - 1, 3000000000, 2 ** 63 - 1, -2 ** 63]: ++ output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) ++ assert x == output ++ ++ def test_int_wire_bytes_at_int32_boundary(self): ++ # The type code stays Int (0x01) inside the Java int range and becomes Long (0x02) ++ # immediately outside it, so promotion never widens a value that already fits. ++ cases = { ++ 2 ** 31 - 1: b'\x01\x00\x7f\xff\xff\xff', ++ -2 ** 31: b'\x01\x00\x80\x00\x00\x00', ++ 2 ** 31: b'\x02\x00\x00\x00\x00\x00\x80\x00\x00\x00', ++ -2 ** 31 - 1: b'\x02\x00\xff\xff\xff\xff\x7f\xff\xff\xff', ++ } ++ for value, expected in cases.items(): ++ assert bytes(self.graphbinary_writer.write_object(value)) == expected ++ ++ def test_int_out_of_int64_range_still_raises(self): ++ # Promotion stops at Long. bigint stays an explicit choice. ++ for x in [2 ** 63, -2 ** 63 - 1]: ++ with pytest.raises(Exception, match='Value too big'): ++ self.graphbinary_writer.write_object(x) ++ ++ def test_int_out_of_int32_range_nested(self): ++ # The reported failure arrived as server-assigned ids inside nested bindings. ++ for x in [[[3000000000, 5000000000]], {'id': 3000000000}]: ++ output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) ++ assert x == output ++ ++ def test_bool_is_not_affected_by_int_promotion(self): ++ # bool subclasses int, so it must keep resolving to the Boolean serializer. ++ for x in [True, False]: ++ output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) ++ assert x == output ++ assert isinstance(output, bool) ++ + def test_long(self): + x = long(100) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_bigint(self): + x = bigint(0x1000_0000_0000_0000_0000) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_bigint_boundaries(self): + # Boundary values around signed byte-width transitions, including + # negative boundaries that previously raised OverflowError (TINKERPOP-3275). + # One representative per equivalence class of the length formula: zero, + # positive byte-width boundary (127/128), negative power-of-two (-128), + # negative just past a boundary (-129/-255, previously OverflowError), the + # obj+1==0 edge (-1), and arbitrary-precision values of both signs. + values = [0, 1, -1, 127, 128, -128, -129, -255, + 0x1000_0000_0000_0000_0000, -0x1000_0000_0000_0000_0000] + for v in values: + x = bigint(v) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_bigint_wire_bytes(self): + # Lock in exact minimal signed two's-complement bytes, matching the Java + # reference BigInteger.toByteArray() (guards against non-minimal encoding). + from gremlin_python.structure.io.graphbinaryV4 import BigIntIO + cases = { + 0: b'\x00\x00\x00\x01\x00', + 127: b'\x00\x00\x00\x01\x7f', + 128: b'\x00\x00\x00\x02\x00\x80', + -128: b'\x00\x00\x00\x01\x80', + -129: b'\x00\x00\x00\x02\xff\x7f', + -256: b'\x00\x00\x00\x02\xff\x00', + } + for value, expected in cases.items(): + assert bytes(BigIntIO.write_bigint(value, bytearray())) == expected + + def test_float(self): + x = float(100.001) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = float('nan') + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert math.isnan(output) + + x = float('-inf') + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert math.isinf(output) and output < 0 + + x = float('inf') + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert math.isinf(output) and output > 0 + + def test_double(self): + x = 100.001 + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_bigdecimal(self): + x = BigDecimal(100, 234) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x.scale == output.scale + assert x.unscaled_value == output.unscaled_value + + def test_bigdecimal_negative_boundaries(self): + # Negative unscaled values at byte-width boundaries previously raised + # OverflowError during serialization (TINKERPOP-3275). + for x in [BigDecimal(0, -129), BigDecimal(2, -255)]: + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x.scale == output.scale + assert x.unscaled_value == output.unscaled_value + + def test_datetime(self): + tz = timezone(timedelta(seconds=36000)) + ms = 12345678912 + x = datetime(2022, 5, 20, tzinfo=tz) + timedelta(microseconds=ms) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_datetime_format(self): + x = datetime.strptime('2022-05-20T03:25:45.678912Z', '%Y-%m-%dT%H:%M:%S.%f%z') + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_datetime_local(self): + x = datetime.now().astimezone() + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_datetime_epoch(self): + x = datetime.fromtimestamp(1690934400).astimezone(timezone.utc) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_string(self): + x = "serialize this!" + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_homogeneous_list(self): + x = ["serialize this!", "serialize that!", "serialize that!","stop telling me what to serialize"] + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_heterogeneous_list(self): + x = ["serialize this!", 0, "serialize that!", "serialize that!", 1, "stop telling me what to serialize", 2] + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_heterogeneous_list_with_none(self): + x = ["serialize this!", 0, "serialize that!", "serialize that!", 1, "stop telling me what to serialize", 2, None] + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_homogeneous_set(self): + x = {"serialize this!", "serialize that!", "stop telling me what to serialize"} + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_heterogeneous_set(self): + x = {"serialize this!", 0, "serialize that!", 1, "stop telling me what to serialize", 2} + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_set_with_unhashable_dict_elements(self): + # test that sets containing dicts are coerced to list - see TINKERPOP-3232 + x = [{"name": "marko", "age": 29}, {"name": "josh", "age": 32}] + list_payload = self.graphbinary_writer.write_object(x) + # patch outer type from list (0x09) to set (0x0b) + set_payload = bytearray(list_payload) + set_payload[0] = 0x0b + output = self.graphbinary_reader.read_object(set_payload) + assert isinstance(output, list) + assert len(output) == 2 + + def test_set_with_unhashable_list_elements(self): + # test that sets containing lists are coerced to list - see TINKERPOP-3232 + list_payload = self.graphbinary_writer.write_object([["marko", "josh"], ["vadas", "peter"]]) + # the first byte is the DataType for list (0x09), change it to set (0x0b) + set_payload = bytearray(list_payload) + set_payload[0] = 0x0b + output = self.graphbinary_reader.read_object(set_payload) + assert isinstance(output, list) + assert len(output) == 2 + + def test_set_with_mixed_hashable_and_unhashable_elements(self): + # test that sets containing a mix of hashable and unhashable elements are coerced to list - see TINKERPOP-3232 + x = ["marko", {"name": "josh"}, 42] + list_payload = self.graphbinary_writer.write_object(x) + set_payload = bytearray(list_payload) + set_payload[0] = 0x0b + output = self.graphbinary_reader.read_object(set_payload) + assert isinstance(output, list) + assert len(output) == 3 + + def test_set_with_nested_unhashable_elements(self): + # test that sets containing dicts with list values are coerced to list - see TINKERPOP-3232 + x = [{"name": "marko", "langs": ["java", "python"]}, {"name": "josh", "langs": ["gremlin"]}] + list_payload = self.graphbinary_writer.write_object(x) + # patch outer type from list (0x09) to set (0x0b) + set_payload = bytearray(list_payload) + set_payload[0] = 0x0b + output = self.graphbinary_reader.read_object(set_payload) + assert isinstance(output, list) + assert len(output) == 2 + + def test_dict(self): + x = {"yo": "what?", + "go": "no!", + "number": 123, + 321: "crazy with the number for a key", + 987: ["go", "deep", {"here": "!"}]} + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = {"marko": [666], "noone": ["blah"]} + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = {"ripple": [], "peter": ["created"], "noone": ["blah"], "vadas": [], + "josh": ["created", "created"], "lop": [], "marko": [666, "created", "knows", "knows"]} + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_ordered_dict(self): + x = OrderedDict() + x['a'] = 1 + x['b'] = 2 + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_uuid(self): + x = uuid.UUID("41d2e28a-20a4-4ab0-b379-d810dede3786") + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_edge(self): + x = Edge(123, Vertex(1, 'person'), "developed", Vertex(10, "software")) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + assert x.inV == output.inV + assert x.outV == output.outV + + def test_edge_with_multi_labelled_endpoints(self): + # the edge itself remains single-labelled, but its endpoint vertices may carry + # multiple labels - see TINKERPOP-3261 + in_vertex = Vertex(1, labels=["person", "employee"]) + out_vertex = Vertex(10, labels=["software", "product"]) + x = Edge(123, out_vertex, "developed", in_vertex) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + assert output.inV.labels == frozenset(["person", "employee"]) + assert output.outV.labels == frozenset(["software", "product"]) + + def test_edge_with_zero_label_endpoints(self): + # endpoint vertices may carry no labels at all under ZERO_OR_MORE cardinality - see TINKERPOP-3261 + in_vertex = Vertex(1, labels=[]) + out_vertex = Vertex(10, labels=[]) + x = Edge(123, out_vertex, "developed", in_vertex) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + assert output.inV.labels == frozenset() + assert output.outV.labels == frozenset() + assert output.inV.label == "" + assert output.outV.label == "" + + def test_path(self): + x = Path(["x", "y", "z"], [1, 2, 3]) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_property(self): + x = Property("name", "stephen", None) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_vertex(self): + x = Vertex(123, "person") + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_vertex_with_multiple_labels(self): + # vertices may carry multiple labels under ONE_OR_MORE/ZERO_OR_MORE cardinality - see TINKERPOP-3261 + x = Vertex(123, labels=["person", "employee"]) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + assert output.labels == frozenset(["person", "employee"]) + + def test_vertex_with_no_labels(self): + # vertices may carry no labels under ZERO_OR_MORE cardinality - see TINKERPOP-3261 + x = Vertex(123, labels=[]) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + assert output.labels == frozenset() + assert output.label == "" + + def test_vertexproperty(self): + x = VertexProperty(123, "name", "stephen", None) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_direction(self): + x = Direction.OUT + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = Direction.from_ + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_byte(self): + x = int.__new__(SingleByte, 1) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_binary(self): + x = bytes("some bytes for you", "utf8") + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_boolean(self): + x = True + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = False + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_char(self): + x = str.__new__(SingleChar, chr(76)) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + x = str.__new__(SingleChar, chr(57344)) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_duration(self): + x = timedelta(seconds=1000, microseconds=1000) + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_marker(self): + x = Marker.end_of_stream() + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(x)) + assert x == output + + def test_graph(self): + graph = Graph() + v1 = Vertex(1, "person") + v2 = Vertex(2, "person") + graph.vertices[1] = v1 + graph.vertices[2] = v2 + e1 = Edge(3, v1, "knows", v2) + graph.edges[3] = e1 + + # Add some properties + vp1 = VertexProperty(4, "name", "marko", v1) + v1.properties.append(vp1) + vp1.properties.append(Property("acl", "public", vp1)) + + e1.properties.append(Property("weight", 0.5, e1)) + + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(graph)) + + assert isinstance(output, Graph) + assert len(output.vertices) == 2 + assert len(output.edges) == 1 + + rv1 = output.vertices[1] + assert rv1.label == "person" + assert len(rv1.properties) == 1 + rvp1 = rv1.properties[0] + assert rvp1.value == "marko" + assert len(rvp1.properties) == 1 + assert rvp1.properties[0].key == "acl" + assert rvp1.properties[0].value == "public" + + re1 = output.edges[3] + assert re1.label == "knows" + assert re1.outV.id == 1 + assert re1.inV.id == 2 + assert len(re1.properties) == 1 + assert re1.properties[0].key == "weight" + assert re1.properties[0].value == 0.5 + + def test_graph_with_multi_labelled_vertices(self): + # a graph containing vertices with multiple labels - see TINKERPOP-3261 + graph = Graph() + graph.vertices[1] = Vertex(1, labels=["person", "employee"]) + graph.vertices[2] = Vertex(2, labels=["software", "product"]) + graph.edges[3] = Edge(3, graph.vertices[1], "created", graph.vertices[2]) + + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(graph)) + + assert isinstance(output, Graph) + assert output.vertices[1].labels == frozenset(["person", "employee"]) + assert output.vertices[2].labels == frozenset(["software", "product"]) + + def test_graph_with_zero_labelled_vertices(self): + # a graph containing vertices with no labels - see TINKERPOP-3261 + graph = Graph() + graph.vertices[1] = Vertex(1, labels=[]) + graph.vertices[2] = Vertex(2, labels=[]) + graph.edges[3] = Edge(3, graph.vertices[1], "created", graph.vertices[2]) + + output = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(graph)) + + assert isinstance(output, Graph) + assert output.vertices[1].labels == frozenset() + assert output.vertices[1].label == "" + assert output.vertices[2].labels == frozenset() + assert output.vertices[2].label == "" + + def test_provider_defined_type(self): + pdt = CompositePDT('Point', {'x': 1, 'y': 2}) + result = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(pdt)) + assert isinstance(result, CompositePDT) + assert result.name == 'Point' + assert result.fields == {'x': 1, 'y': 2} + + def test_provider_defined_type_nested(self): + inner = CompositePDT('Address', {'street': 'Main'}) + outer = CompositePDT('Person', {'name': 'Alice', 'address': inner}) + result = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(outer)) + assert result.name == 'Person' + assert result.fields['name'] == 'Alice' + assert isinstance(result.fields['address'], CompositePDT) + assert result.fields['address'].name == 'Address' + + def test_provider_defined_type_null_field(self): + pdt = CompositePDT('NullableType', {'value': None, 'name': 'test'}) + result = self.graphbinary_reader.read_object(self.graphbinary_writer.write_object(pdt)) + assert result.fields['value'] is None + assert result.fields['name'] == 'test'
