Skip to content
2 changes: 1 addition & 1 deletion .github/workflows/test-python-examples.yml
Original file line number Diff line number Diff line change
Expand Up @@ -366,7 +366,7 @@ jobs:
example_jvm_args=""
;;
"11_vector_index_build.py")
example_args="--backend arcadedb_sql --dataset stackoverflow-tiny --threads 1 --mem-limit 2g --batch-size 500 --max-connections 16 --beam-width 100 --quantization NONE --run-label ci11_arcadedb_sql"
example_args="--backend arcadedb_sql --dataset stackoverflow-tiny --threads 1 --mem-limit 2g --batch-size 500 --beam-width 100 --quantization NONE --run-label ci11_arcadedb_sql"
example_name="$example (vector build, arcadedb_sql backend, minimal)"
timeout_duration=1200
example_jvm_args=""
Expand Down
29 changes: 20 additions & 9 deletions bindings/python/examples/03_vector_search.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,15 +220,6 @@ def quantize_to_int8_bytes(vector: np.ndarray):
print("Step 5: Creating vector index...")
step_start = time.time()

print(f" 💡 JVector Parameters:")
print(f" • dimensions: {EMBEDDING_DIM} (matches embedding size)")
print(" • distance_function: cosine (best for normalized vectors)")
print(
" • max_connections: 32 (connections per node, higher = more accurate but slower)"
)
print(" • beam_width: 256 (search quality, higher = more accurate)")
print()

db.command(
"sql",
f"""
Expand All @@ -241,6 +232,26 @@ def quantize_to_int8_bytes(vector: np.ndarray):
""",
)

# Report what the index was actually built with. The METADATA above sets only
# dimensions and similarity, so everything else comes from the engine defaults;
# reading them back keeps this output correct if those defaults change.
index_meta = db.schema.get_vector_index("Article", "embedding").get_metadata()
print(" 💡 JVector Parameters:")
print(f" • dimensions: {index_meta['dimensions']} (matches embedding size)")
print(
f" • distance_function: {index_meta['similarity_function']}"
" (best for normalized vectors)"
)
print(
f" • max_connections: {index_meta['max_connections']}"
" (connections per node, higher = more accurate but slower)"
)
print(
f" • beam_width: {index_meta['beam_width']}"
" (search quality, higher = more accurate)"
)
print()

print(" ✅ Created JVector vector index")
print(" ✅ Built vector index graph immediately via SQL")
print(" 💡 LSM index automatically indexes existing records upon creation.")
Expand Down
10 changes: 9 additions & 1 deletion bindings/python/examples/06_vector_search_recommendations.py
Original file line number Diff line number Diff line change
Expand Up @@ -250,7 +250,6 @@ def create_sql_vector_index(db, property_suffix=""):
num_movies = len(result_list)

print(f"\nCreating HNSW (JVector) index for {embedding_prop}...")
print(" metric=cosine, max_connections=32, beam_width=256")

start_time = time.time()

Expand All @@ -267,6 +266,15 @@ def create_sql_vector_index(db, property_suffix=""):
)

elapsed = time.time() - start_time

# The METADATA above sets only dimensions and similarity, so the rest comes from
# the engine defaults; read them back rather than restating them here.
index_meta = db.schema.get_vector_index("Movie", embedding_prop).get_metadata()
print(
f" metric={index_meta['similarity_function']},"
f" max_connections={index_meta['max_connections']},"
f" beam_width={index_meta['beam_width']} (engine defaults)"
)
print(f"✓ Created and indexed {num_movies:,} movies in {elapsed:.1f}s")


Expand Down
9 changes: 6 additions & 3 deletions bindings/python/examples/09_stackoverflow_graph_oltp.py
Original file line number Diff line number Diff line change
Expand Up @@ -297,7 +297,7 @@ def get_arcadedb_module():

def get_ladybug_module():
try:
import real_ladybug as lb
import ladybug as lb
except ImportError:
return None
return lb
Expand Down Expand Up @@ -5397,7 +5397,7 @@ def run_graph_oltp_ladybug(
) -> dict:
lb = get_ladybug_module()
if lb is None:
raise RuntimeError("real_ladybug is not installed")
raise RuntimeError("ladybug is not installed")

if db_path.exists():
shutil.rmtree(db_path)
Expand Down Expand Up @@ -9724,7 +9724,10 @@ def run_in_docker(args) -> bool:
else:
packages = ["lxml"]
if args.db in ("ladybug", "ladybugdb"):
packages.append("real_ladybug")
# Pinned for reproducible benchmark numbers: ladybug is pre-1.0 and no CI
# job exercises this path, so an unpinned minor bump would go unnoticed.
# Re-check against the latest release when refreshing published results.
packages.append("ladybug==0.19.0")
if args.db == "graphqlite":
packages.append("graphqlite")
if args.db == "duckdb":
Expand Down
9 changes: 6 additions & 3 deletions bindings/python/examples/10_stackoverflow_graph_olap.py
Original file line number Diff line number Diff line change
Expand Up @@ -362,7 +362,7 @@ def get_arcadedb_module():

def get_ladybug_module():
try:
import real_ladybug as lb
import ladybug as lb
except ImportError:
return None
return lb
Expand Down Expand Up @@ -4336,7 +4336,7 @@ def run_olap_ladybug(
) -> dict:
lb = get_ladybug_module()
if lb is None:
raise RuntimeError("real_ladybug is not installed")
raise RuntimeError("ladybug is not installed")

if db_path.exists():
shutil.rmtree(db_path)
Expand Down Expand Up @@ -6566,7 +6566,10 @@ def run_in_docker(args) -> bool:
else:
packages = ["lxml"]
if args.db in ("ladybug", "ladybugdb"):
packages.append("real_ladybug")
# Pinned for reproducible benchmark numbers: ladybug is pre-1.0 and no CI
# job exercises this path, so an unpinned minor bump would go unnoticed.
# Re-check against the latest release when refreshing published results.
packages.append("ladybug==0.19.0")
if args.db == "graphqlite":
packages.append("graphqlite")
if args.db == "duckdb":
Expand Down
40 changes: 28 additions & 12 deletions bindings/python/examples/11_vector_index_build.py
Original file line number Diff line number Diff line change
Expand Up @@ -609,12 +609,23 @@
)


def hnsw_m_from_max_connections(max_connections) -> int:
"""Convert an ArcadeDB per-layer degree into the equivalent hnswlib M.

Check notice on line 613 in bindings/python/examples/11_vector_index_build.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/examples/11_vector_index_build.py#L613

Multi-line docstring summary should start at the second line (D213)

ArcadeDB applies maxConnections verbatim to every layer including the base
layer. hnswlib-derived backends allocate 2*M links at the base layer, so an
M of half the ArcadeDB degree produces the same base-layer density and keeps
the comparison degree-matched.
"""
return max(1, int(max_connections) // 2)


def create_index_faiss(dim: int, max_connections: int, beam_width: int):
import faiss

index_hnsw = faiss.IndexHNSWFlat(
int(dim),
int(max_connections),
hnsw_m_from_max_connections(max_connections),
faiss.METRIC_INNER_PRODUCT,
)
index_hnsw.hnsw.efConstruction = int(beam_width)
Expand Down Expand Up @@ -693,10 +704,11 @@
max_connections: int,
beam_width: int,
) -> dict:
hnsw_m = hnsw_m_from_max_connections(max_connections)
common_kwargs = {
"metric": "cosine",
"vector_column_name": "vector",
"m": int(max_connections),
"m": hnsw_m,
"ef_construction": int(beam_width),
}
attempts = [
Expand All @@ -715,7 +727,7 @@
config = {
"metric": "cosine",
"index_type": index_type,
"hnsw_m": int(max_connections),
"hnsw_m": hnsw_m,
"hnsw_ef_construct": int(beam_width),
}
if "num_partitions" in extra_kwargs:
Expand Down Expand Up @@ -840,7 +852,7 @@


def create_index_pgvector(conn, max_connections: int, beam_width: int) -> None:
m_val = int(max_connections)
m_val = hnsw_m_from_max_connections(max_connections)
ef_val = int(beam_width)
with conn.cursor() as cur:
cur.execute(
Expand All @@ -864,7 +876,7 @@
distance=models.Distance.COSINE,
),
hnsw_config=models.HnswConfigDiff(
m=int(max_connections),
m=hnsw_m_from_max_connections(max_connections),
ef_construct=int(beam_width),
),
)
Expand Down Expand Up @@ -1296,7 +1308,7 @@
"index_type": "HNSW",
"metric_type": "COSINE",
"params": {
"M": int(max_connections),
"M": hnsw_m_from_max_connections(max_connections),
"efConstruction": int(beam_width),
},
}
Expand Down Expand Up @@ -1884,8 +1896,12 @@
parser.add_argument(
"--max-connections",
type=int,
default=16,
help="HNSW max connections / m (default: 16)",
default=32,
help=(
"ArcadeDB per-layer graph degree (default: 32). hnswlib-derived "
"backends receive half this value as M, so every backend builds at "
"the same base-layer density"
),
)
parser.add_argument(
"--beam-width",
Expand Down Expand Up @@ -2636,7 +2652,7 @@
"qdrant": {
"data_dir": str(db_path / "qdrant-data"),
"collection": "vectordata",
"hnsw_m": args.max_connections,
"hnsw_m": hnsw_m_from_max_connections(args.max_connections),
"hnsw_ef_construct": args.beam_width,
},
"milvus": {
Expand All @@ -2645,13 +2661,13 @@
"compose_version": args.milvus_compose_version,
"compose_file": str(db_path / "milvus-compose" / "docker-compose.yml"),
"collection": args.milvus_collection,
"hnsw_m": args.max_connections,
"hnsw_m": hnsw_m_from_max_connections(args.max_connections),
"hnsw_ef_construct": args.beam_width,
},
"faiss": {
"index_file": str(db_path / "faiss.index"),
"metric": "cosine_via_inner_product_normalized",
"hnsw_m": args.max_connections,
"hnsw_m": hnsw_m_from_max_connections(args.max_connections),
"hnsw_ef_construct": args.beam_width,
},
"lancedb": {
Expand All @@ -2669,7 +2685,7 @@
),
"num_partitions": ((lancedb_index_config or {}).get("num_partitions")),
"quantization": ((lancedb_index_config or {}).get("quantization")),
"hnsw_m": args.max_connections,
"hnsw_m": hnsw_m_from_max_connections(args.max_connections),
"hnsw_ef_construct": args.beam_width,
},
},
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1328,7 +1328,7 @@ def create_sql_vector_index(db, vertex_type: str) -> float:
METADATA {{
"dimensions": 384,
"similarity": "COSINE",
"maxConnections": 16,
"maxConnections": 32,
"beamWidth": 100,
"quantization": "INT8",
"storeVectorsInGraph": false,
Expand Down
9 changes: 6 additions & 3 deletions bindings/python/src/arcadedb_embedded/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -338,7 +338,7 @@ def create_vector_index(
dimensions: int,
id_property: Optional[str] = None,
distance_function: str = "cosine",
max_connections: int = 16,
max_connections: int = 32,
beam_width: int = 100,
quantization: str = "INT8",
encoding: Optional[str] = None,
Expand Down Expand Up @@ -370,8 +370,11 @@ def create_vector_index(
id_property: Optional property used for key-based vector lookup.
Defaults to the engine default (usually "id") when omitted.
distance_function: "cosine", "euclidean", or "inner_product"
max_connections: Max connections per node (default: 16).
Maps to `maxConnections` in JVector.
max_connections: Per-layer graph degree (default: 32, matching the
engine default since #5352). Maps to `maxConnections` in
JVector, which is a Vamana per-layer degree and is NOT doubled
at the base layer like hnswlib's M: to reproduce an
hnswlib-style configuration use max_connections = 2 * M.
beam_width: Beam width for search/construction (default: 100).
Maps to `beamWidth` in JVector.
quantization: Vector quantization type (default: INT8).
Expand Down
45 changes: 45 additions & 0 deletions bindings/python/tests/test_example11_degree_matching.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
"""Example 11 compares ArcadeDB against hnswlib-derived vector backends.

Check notice on line 1 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L1

Multi-line docstring summary should start at the second line (D213)

ArcadeDB applies maxConnections verbatim to every graph layer. hnswlib-derived
indexes (faiss, lancedb, pgvector, qdrant, milvus) allocate 2*M links at the
base layer. Passing one number to both families unchanged makes the benchmark
compare different graph densities, so the example converts between them.
"""

from __future__ import annotations

import importlib.util
from pathlib import Path

import pytest

EXAMPLE_PATH = (
Path(__file__).resolve().parents[1] / "examples" / "11_vector_index_build.py"
)

pytestmark = pytest.mark.skipif(
not EXAMPLE_PATH.exists(),
reason="bindings/python/examples/11_vector_index_build.py is not present",
)


@pytest.fixture(scope="module")
def example11():
spec = importlib.util.spec_from_file_location("example11", EXAMPLE_PATH)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module


def test_hnsw_m_is_half_the_vamana_degree(example11):
assert example11.hnsw_m_from_max_connections(32) == 16

Check warning on line 35 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L35

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.
assert example11.hnsw_m_from_max_connections(64) == 32

Check warning on line 36 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L36

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.


def test_hnsw_m_never_drops_below_one(example11):
assert example11.hnsw_m_from_max_connections(1) == 1

Check warning on line 40 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L40

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.
assert example11.hnsw_m_from_max_connections(0) == 1

Check warning on line 41 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L41

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.


def test_hnsw_m_accepts_string_input(example11):
assert example11.hnsw_m_from_max_connections("32") == 16

Check warning on line 45 in bindings/python/tests/test_example11_degree_matching.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_example11_degree_matching.py#L45

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.
45 changes: 45 additions & 0 deletions bindings/python/tests/test_vector_params_verification.py
Original file line number Diff line number Diff line change
Expand Up @@ -296,6 +296,51 @@

assert str(idx_to_check.getMetadata().quantizationType) == "PRODUCT"

@staticmethod
def _engine_default_max_connections() -> int:
"""Read maxConnections off a freshly constructed engine metadata object.

Check notice on line 301 in bindings/python/tests/test_vector_params_verification.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_vector_params_verification.py#L301

Multi-line docstring summary should start at the second line (D213)

The field initializer is the engine default, so the value tracks the
bundled jars rather than a number copied into the Python layer.
"""
import jpype

java_string = jpype.JClass("java.lang.String")
property_names = jpype.JArray(java_string)(["embedding"])
metadata = jpype.JPackage("com").arcadedb.schema.LSMVectorIndexMetadata(
"DefaultProbe", property_names, 0
)
return int(metadata.maxConnections)

def test_wrapper_default_matches_engine_default(self, test_db):
"""The Python default must equal the engine default, not a copy of it.

Check notice on line 316 in bindings/python/tests/test_vector_params_verification.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_vector_params_verification.py#L316

Multi-line docstring summary should start at the second line (D213)

create_vector_index always calls withMaxConnections, so the wrapper
default fully shadows the engine default for callers who omit it.
"""
import inspect

signature = inspect.signature(type(test_db).create_vector_index)
wrapper_default = signature.parameters["max_connections"].default

assert wrapper_default == self._engine_default_max_connections()

Check warning on line 326 in bindings/python/tests/test_vector_params_verification.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_vector_params_verification.py#L326

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.

def test_omitted_max_connections_reaches_the_index(self, test_db):
"""Omitting the argument must land the engine default on the index."""
engine_default = self._engine_default_max_connections()

test_db.command("sql", "CREATE VERTEX TYPE DefaultDegreeDoc")
test_db.command(
"sql", "CREATE PROPERTY DefaultDegreeDoc.embedding ARRAY_OF_FLOATS"
)

index = test_db.create_vector_index(
"DefaultDegreeDoc", "embedding", dimensions=4
)

metadata = self._get_primary_metadata(index)
assert int(metadata.maxConnections) == engine_default

Check warning on line 342 in bindings/python/tests/test_vector_params_verification.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

bindings/python/tests/test_vector_params_verification.py#L342

Use of assert detected. The enclosed code will be removed when compiling to optimised byte code.

def test_jvm_heap_check(self):
"""Verify JVM memory settings from Java level."""
import jpype
Expand Down
Loading