Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
/*
* Copyright DataStax, Inc.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package io.github.jbellis.jvector.example.benchmarks.datasets;

import java.nio.file.Path;

class DataSetFiles {
private final Path baseFvecsPath;
private final Path queryFvecsPath;
private final Path gtIvecsPath;

DataSetFiles(Path baseFvecsPath, Path queryFvecsPath, Path gtIvecsPath) {
this.baseFvecsPath = baseFvecsPath;
this.queryFvecsPath = queryFvecsPath;
this.gtIvecsPath = gtIvecsPath;
}

Path getBaseFvecsPath() {
return baseFvecsPath;
}

Path getQueryFvecsPath() {
return queryFvecsPath;
}

Path getGtIvecsPath() {
return gtIvecsPath;
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -16,122 +16,12 @@

package io.github.jbellis.jvector.example.benchmarks.datasets;

import io.github.jbellis.jvector.vector.VectorSimilarityFunction;
import java.util.Optional;
import java.util.function.Supplier;

/// A lightweight, lazy handle that separates *identifying* a dataset from *loading* its data.
///
/// Metadata such as the dataset name and similarity function are available immediately
/// without any I/O, while the expensive work of reading vectors, deduplicating, scrubbing
/// zero vectors, and normalizing is deferred until the first call to {@link #getDataSet()}.
///
/// This design allows callers to enumerate or filter available datasets cheaply, and
/// ensures that the full load-and-scrub pipeline runs at most once per handle thanks to
/// thread-safe caching.
///
/// Instances are created by {@link DataSetLoader} implementations; callers obtain them
/// through {@link DataSets#loadDataSet(String)}.
///
/// ### Typical usage
/// ```java
/// DataSetInfo info = DataSets.loadDataSet("ada002-100k").orElseThrow();
///
/// // Cheap — no vectors loaded yet
/// System.out.println(info.getName());
/// System.out.println(info.similarityFunction());
///
/// // First call triggers full load; subsequent calls return the cached DataSet
/// DataSet ds = info.getDataSet();
/// ```
///
/// @see DataSet
/// @see DataSetLoader
/// @see DataSets
public class DataSetInfo implements DataSetProperties {
private final Supplier<DataSet> loader;
private final DataSetProperties baseProperties;
private volatile DataSet cached;

/// Creates a new dataset info handle.
///
/// The supplied {@code loader} will not be invoked until {@link #getDataSet()} is called.
/// It should perform the full load-and-scrub pipeline (read vectors, remove duplicates /
/// zero vectors, filter queries, normalize) and return a ready-to-use {@link DataSet}.
///
/// @param baseProperties the dataset properties (name, similarity function, etc.)
/// @param loader a supplier that performs the deferred load; invoked at most once
public DataSetInfo(DataSetProperties baseProperties, Supplier<DataSet> loader) {
this.baseProperties = baseProperties;
this.loader = loader;
}

/**
* {@inheritDoc}
*/
@Override
public Optional<VectorSimilarityFunction> similarityFunction() {
return baseProperties.similarityFunction();
}

/**
* {@inheritDoc}
*/
@Override
public int numVectors() {
return this.baseProperties.numVectors();
}

/**
* {@inheritDoc}
*/
@Override
public String getName() {
return baseProperties.getName();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isNormalized() {
return baseProperties.isNormalized();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isZeroVectorFree() {
return baseProperties.isZeroVectorFree();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isDuplicateVectorFree() {
return baseProperties.isDuplicateVectorFree();
}

/// Returns the fully loaded and scrubbed {@link DataSet}.
///
/// On the first invocation this triggers the deferred load pipeline, which may involve
/// reading large vector files from disk, deduplication, zero-vector removal, and
/// normalization. The result is cached so that subsequent calls return immediately.
public interface DataSetInfo extends DataSetProperties {
/// Loads and returns a {@link DataSet} corresponding to the underlying source.
///
/// This method is thread-safe: concurrent callers will block until the first load
/// completes, after which all callers share the same cached instance.
/// This method may incur an IO penalty based on the size of the dataset and it's source.
/// Implementations are not required to cache the dataset or ensure thread-safety.
///
/// @return the ready-to-use {@link DataSet}
public DataSet getDataSet() {
if (cached == null) {
synchronized (this) {
if (cached == null) {
cached = loader.get();
}
}
}
return cached;
}
public DataSet getDataSet();
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,139 @@
/*
* Copyright DataStax, Inc.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package io.github.jbellis.jvector.example.benchmarks.datasets;

import io.github.jbellis.jvector.example.util.SiftLoader;
import io.github.jbellis.jvector.vector.VectorSimilarityFunction;

import java.util.Optional;

/// A lightweight, lazy handle that separates *identifying* a dataset from *loading* its data.
///
/// Metadata such as the dataset name and similarity function are available immediately
/// without any I/O, while the expensive work of reading vectors, deduplicating, scrubbing
/// zero vectors, and normalizing is deferred until the first call to {@link #getDataSet()}.
///
/// This design allows callers to enumerate or filter available datasets cheaply, and
/// ensures that the full load-and-scrub pipeline runs at most once per handle thanks to
/// thread-safe caching.
///
/// Instances are created by {@link DataSetLoader} implementations; callers obtain them
/// through {@link DataSets#loadDataSet(String)}.
///
/// ### Typical usage
/// ```java
/// DataSetInfo info = DataSets.loadDataSet("ada002-100k").orElseThrow();
///
/// // Cheap — no vectors loaded yet
/// System.out.println(info.getName());
/// System.out.println(info.similarityFunction());
///
/// // First call triggers full load; subsequent calls return the cached DataSet
/// DataSet ds = info.getDataSet();
/// ```
///
/// @see DataSet
/// @see DataSetLoader
/// @see DataSets
public class DataSetInfoMFD implements DataSetInfo {
private final DataSetFiles dsFiles;
private final DataSetProperties baseProperties;
private volatile DataSet cached;

/// Creates a new dataset info handle.
///
/// The dataset will not be loaded until {@link #getDataSet()} is called for the first time.
///
/// @param baseProperties the dataset properties (name, similarity function, etc.)
/// @param dataSetFiles the bundle of base/query/gt paths required to load the dataset
public DataSetInfoMFD(DataSetProperties baseProperties, DataSetFiles dataSetFiles) {
this.baseProperties = baseProperties;
this.dsFiles = dataSetFiles;
}

/**
* {@inheritDoc}
*/
@Override
public Optional<VectorSimilarityFunction> similarityFunction() {
return baseProperties.similarityFunction();
}

/**
* {@inheritDoc}
*/
@Override
public int numVectors() {
return this.baseProperties.numVectors();
}

/**
* {@inheritDoc}
*/
@Override
public String getName() {
return baseProperties.getName();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isNormalized() {
return baseProperties.isNormalized();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isZeroVectorFree() {
return baseProperties.isZeroVectorFree();
}

/**
* {@inheritDoc}
*/
@Override
public boolean isDuplicateVectorFree() {
return baseProperties.isDuplicateVectorFree();
}

/// Returns the fully loaded and scrubbed {@link DataSet}.
///
/// On the first invocation this triggers the deferred load pipeline, which may involve
/// reading large vector files from disk, deduplication, zero-vector removal, and
/// normalization. The result is cached so that subsequent calls return immediately.
///
/// This method is thread-safe: concurrent callers will block until the first load
/// completes, after which all callers share the same cached instance.
///
/// @return the ready-to-use {@link DataSet}

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Datasets should never be scrubbed online. Is this comment out of date?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

No, the comment is still accurate, and preserves the existing behavior. DataSets with the LEGACY_SCRUB config continue to be scrubbed on load (by processDataSet)

public DataSet getDataSet() {
if (cached == null) {
synchronized (this) {
if (cached == null) {
var baseVectors = SiftLoader.readFvecs(dsFiles.getBaseFvecsPath().toString());
var queryVectors = SiftLoader.readFvecs(dsFiles.getQueryFvecsPath().toString());
var gtVectors = SiftLoader.readIvecs(dsFiles.getGtIvecsPath().toString());
cached = DataSetUtils.processDataSet(baseProperties.getName(), baseProperties, baseVectors, queryVectors, gtVectors);
}
}
}
return cached;
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -46,5 +46,5 @@ public interface DataSetLoader {
* @param dataSetName the logical dataset name (not a filename; do not include extensions like {@code .hdf5})
* @return a {@link DataSetInfo} handle for the dataset, if found
*/
Optional<DataSetInfo> loadDataSet(String dataSetName);
Optional<? extends DataSetInfo> loadDataSet(String dataSetName);
}
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,6 @@
*/
package io.github.jbellis.jvector.example.benchmarks.datasets;

import io.github.jbellis.jvector.example.util.SiftLoader;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import org.yaml.snakeyaml.Yaml;
Expand Down Expand Up @@ -386,7 +385,7 @@ public DataSetLoaderSimpleMFD(String catalogUrl, String localPath, boolean check
}

@Override
public Optional<DataSetInfo> loadDataSet(String dataSetName) {
public Optional<DataSetInfoMFD> loadDataSet(String dataSetName) {
var entry = catalog.get(dataSetName);
if (entry == null) return Optional.empty();

Expand Down Expand Up @@ -423,12 +422,11 @@ public Optional<DataSetInfo> loadDataSet(String dataSetName) {
String.format(
"Dataset '%s' was found in dataset catalog, but no metadata entry was found in dataset-metadata.yml. ",
dataSetName)));
return Optional.of(new DataSetInfo(props, () -> {
var baseVectors = SiftLoader.readFvecs(effectiveCacheDir.resolve(baseFile).toString());
var queryVectors = SiftLoader.readFvecs(effectiveCacheDir.resolve(queryFile).toString());
var gtVectors = SiftLoader.readIvecs(effectiveCacheDir.resolve(gtFile).toString());
return DataSetUtils.processDataSet(dataSetName, props, baseVectors, queryVectors, gtVectors);
}));
return Optional.of(new DataSetInfoMFD(props, new DataSetFiles(
effectiveCacheDir.resolve(baseFile),
effectiveCacheDir.resolve(queryFile),
effectiveCacheDir.resolve(gtFile)
)));
}

// ========================================================================================
Expand Down Expand Up @@ -785,7 +783,6 @@ private void saveCatalogLocally(Path localCatalog, String catalogUrl,
}
}

@SuppressWarnings("unchecked")
private static Map<String, Map<String, String>> loadCatalogFromFile(Path path) {
try (InputStream in = Files.newInputStream(path)) {
Map<String, Map<String, String>> result = new Yaml().load(in);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,7 @@ public class DataSets {
///
/// @param dataSetName the logical dataset name (e.g. {@code "ada002-100k"})
/// @return a lazy {@link DataSetInfo} handle, or empty if no loader recognises the name
public static Optional<DataSetInfo> loadDataSet(String dataSetName) {
public static Optional<? extends DataSetInfo> loadDataSet(String dataSetName) {
return loadDataSet(dataSetName, defaultLoaders);
}

Expand All @@ -61,15 +61,15 @@ public static Optional<DataSetInfo> loadDataSet(String dataSetName) {
/// @param dataSetName the logical dataset name (e.g. {@code "ada002-100k"})
/// @param loaders the loaders to try, in priority order
/// @return a lazy {@link DataSetInfo} handle, or empty if no loader recognises the name
public static Optional<DataSetInfo> loadDataSet(String dataSetName, Collection<DataSetLoader> loaders) {
public static Optional<? extends DataSetInfo> loadDataSet(String dataSetName, Collection<DataSetLoader> loaders) {
logger.info("loading dataset [{}]", dataSetName);
if (dataSetName.endsWith(".hdf5")) {
throw new InvalidParameterException("DataSet names are not meant to be file names. Did you mean " + dataSetName.replace(".hdf5", "") + "? ");
}

for (DataSetLoader loader : loaders) {
logger.trace("trying loader [{}]", loader.getClass().getSimpleName());
Optional<DataSetInfo> dataSetLoaded = loader.loadDataSet(dataSetName);
var dataSetLoaded = loader.loadDataSet(dataSetName);
if (dataSetLoaded.isPresent()) {
logger.info("dataset [{}] found with loader [{}]", dataSetName, loader.getClass().getSimpleName());
return dataSetLoaded;
Expand Down
Loading
Loading