Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
72 changes: 72 additions & 0 deletions core/src/main/java/org/apache/iceberg/ColumnFile.java
Original file line number Diff line number Diff line change
@@ -0,0 +1,72 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one
* or more contributor license agreements. See the NOTICE file
* distributed with this work for additional information
* regarding copyright ownership. The ASF licenses this file
* to you under the Apache License, Version 2.0 (the
* "License"); you may not use this file except in compliance
* with the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing,
* software distributed under the License is distributed on an
* "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
* KIND, either express or implied. See the License for the
* specific language governing permissions and limitations
* under the License.
*/
package org.apache.iceberg;

import java.nio.ByteBuffer;
import java.util.List;
import org.apache.iceberg.types.Types;

interface ColumnFile {
Comment thread
gaborkaszab marked this conversation as resolved.
Types.NestedField LOCATION =
Types.NestedField.required(
160, "location", Types.StringType.get(), "Location of the column file");
Types.NestedField FIELD_IDS =

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

If we keep the field id's here, does that mean we need to update all column file entities whenever we add a new column file? I think that's probably alright but we should probably add some text on that requirement?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yes, when we add a new column file, we have to remove its field IDs from other column files' field ID lists. I'm not sure we should document it here, this just the storage representation of such a structure. The "deduplication" of field IDs happen on a different level, maybe document this on the API where we add the new column files (non-existing as of now)?

Types.NestedField.required(
161,
"field_ids",
Types.ListType.ofRequired(162, Types.IntegerType.get()),
"Live field IDs in this column file");
Types.NestedField FILE_FORMAT =
Types.NestedField.required(
163,
"file_format",
Types.StringType.get(),
"String file format name for this column file");
Types.NestedField FILE_SIZE_IN_BYTES =
Types.NestedField.required(
164, "file_size_in_bytes", Types.LongType.get(), "Total column file size in bytes");
Types.NestedField KEY_METADATA =
Types.NestedField.optional(
165,
"key_metadata",
Types.BinaryType.get(),
"Key metadata for encryption; specific to the encryption scheme.");

static Types.StructType schema() {
return Types.StructType.of(LOCATION, FIELD_IDS, FILE_FORMAT, FILE_SIZE_IN_BYTES, KEY_METADATA);
}

/** Returns the location of this column file. */
String location();

/** Returns the field IDs contained in this column file. */
List<Integer> fieldIds();

/** Returns the format of this column file. */
FileFormat fileFormat();

/** Returns the total size of this column file in bytes. */
long fileSizeInBytes();

/** Returns encryption key metadata, or null if this column file is not encrypted. */
ByteBuffer keyMetadata();

/** Copies this column file. */
ColumnFile copy();
}
218 changes: 218 additions & 0 deletions core/src/main/java/org/apache/iceberg/ColumnFileStruct.java
Original file line number Diff line number Diff line change
@@ -0,0 +1,218 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one
* or more contributor license agreements. See the NOTICE file
* distributed with this work for additional information
* regarding copyright ownership. The ASF licenses this file
* to you under the Apache License, Version 2.0 (the
* "License"); you may not use this file except in compliance
* with the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing,
* software distributed under the License is distributed on an
* "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
* KIND, either express or implied. See the License for the
* specific language governing permissions and limitations
* under the License.
*/
package org.apache.iceberg;

import java.io.Serializable;
import java.nio.ByteBuffer;
import java.util.Arrays;
import java.util.List;
import org.apache.iceberg.avro.SupportsIndexProjection;
import org.apache.iceberg.relocated.com.google.common.base.MoreObjects;
import org.apache.iceberg.relocated.com.google.common.base.Preconditions;
import org.apache.iceberg.relocated.com.google.common.collect.Sets;
import org.apache.iceberg.types.Types;
import org.apache.iceberg.util.ArrayUtil;
import org.apache.iceberg.util.ByteBuffers;

/** Mutable {@link StructLike} implementation of {@link ColumnFile}. */
class ColumnFileStruct extends SupportsIndexProjection implements ColumnFile, Serializable {
private static final Types.StructType BASE_TYPE =
Types.StructType.of(
ColumnFile.LOCATION,
ColumnFile.FIELD_IDS,
ColumnFile.FILE_FORMAT,
ColumnFile.FILE_SIZE_IN_BYTES,
ColumnFile.KEY_METADATA);

private String location = null;
private int[] fieldIds = null;
private FileFormat fileFormat = null;
private long fileSizeInBytes = -1L;
private byte[] keyMetadata = null;

/** Used by internal readers to instantiate this class with a projection schema. */
ColumnFileStruct(Types.StructType projection) {
super(BASE_TYPE, projection);
}

ColumnFileStruct(
String location,
List<Integer> fieldIds,
FileFormat fileFormat,
long fileSizeInBytes,
ByteBuffer keyMetadata) {
super(BASE_TYPE.fields().size());
this.location = location;
this.fieldIds = ArrayUtil.toIntArray(fieldIds);
this.fileFormat = fileFormat;
this.fileSizeInBytes = fileSizeInBytes;
this.keyMetadata = ByteBuffers.toByteArray(keyMetadata);
}

/** Copy constructor. */
private ColumnFileStruct(ColumnFileStruct toCopy) {
super(toCopy);
this.location = toCopy.location;
this.fieldIds =
toCopy.fieldIds != null ? Arrays.copyOf(toCopy.fieldIds, toCopy.fieldIds.length) : null;
this.fileFormat = toCopy.fileFormat;
this.fileSizeInBytes = toCopy.fileSizeInBytes;
this.keyMetadata =
toCopy.keyMetadata != null
? Arrays.copyOf(toCopy.keyMetadata, toCopy.keyMetadata.length)
: null;
}

/** Constructor for Java serialization. */
ColumnFileStruct() {
super(BASE_TYPE.fields().size());
}

@Override
public String location() {
return location;
}

@Override
public List<Integer> fieldIds() {
return fieldIds != null ? ArrayUtil.toUnmodifiableIntList(fieldIds) : null;
}

@Override
public FileFormat fileFormat() {
return fileFormat;
}

@Override
public long fileSizeInBytes() {
return fileSizeInBytes;
}

@Override
public ByteBuffer keyMetadata() {
return keyMetadata != null ? ByteBuffer.wrap(keyMetadata) : null;
}

@Override
public ColumnFile copy() {
return new ColumnFileStruct(this);
}

@Override
protected <T> T internalGet(int pos, Class<T> javaClass) {
return javaClass.cast(getByPos(pos));
}

private Object getByPos(int pos) {
return switch (pos) {
case 0 -> location;
case 1 -> fieldIds();
case 2 -> fileFormat != null ? fileFormat.toString() : null;
case 3 -> fileSizeInBytes;
case 4 -> keyMetadata();
default -> throw new UnsupportedOperationException("Unknown field ordinal: " + pos);
};
}

@Override
@SuppressWarnings("unchecked")
protected <T> void internalSet(int pos, T value) {
switch (pos) {
// always coerce to String for Serializable
Comment thread
stevenzwu marked this conversation as resolved.
case 0 -> this.location = value.toString();
case 1 -> this.fieldIds = ArrayUtil.toIntArray((List<Integer>) value);
case 2 -> this.fileFormat = FileFormat.fromString(value.toString());
case 3 -> this.fileSizeInBytes = (long) value;
case 4 -> this.keyMetadata = ByteBuffers.toByteArray((ByteBuffer) value);
default -> {
// ignore the object, it must be from a newer version of the format

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

nit: should the comment say `ignore the unknown positions, as they must come from a newer version of the format"

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This comment is inline with the same in TrackedFileStruct and TrackingStruct. I'd rather keep consistency with these.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

that's fine for consistency. I found "ignore the object" not very accurate.

}
}
}

static Builder builder() {
return new Builder();
}

@Override
public String toString() {
return MoreObjects.toStringHelper(this)
.add("location", location)
.add("field_ids", fieldIds)
.add("file_format", fileFormat)
.add("file_size_in_bytes", fileSizeInBytes)
.add("key_metadata", keyMetadata == null ? "null" : "(redacted)")
.toString();
}

static class Builder {

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Since ColumnFile is something being passed in by the users through the table API, at some point we would want to make some builder public for them. I'd keep this here as long as it's possible and then I'd introduce something like a ColumnFiles class that can be used to produce the ColumnFile objects, similarly to DataFiles

private String location = null;
private List<Integer> fieldIds = null;
private FileFormat fileFormat = null;
private Long fileSizeInBytes = null;
private ByteBuffer keyMetadata = null;

Builder location(String newLocation) {
Preconditions.checkArgument(newLocation != null, "Invalid location: null");
Preconditions.checkArgument(!newLocation.isEmpty(), "Invalid location: empty");
this.location = newLocation;
return this;
}

Builder fieldIds(List<Integer> newFieldIds) {
Preconditions.checkArgument(newFieldIds != null, "Invalid field IDs: null");
Preconditions.checkArgument(!newFieldIds.isEmpty(), "Invalid field IDs: empty");

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does this mean we have to remove a column file if we override all the columns in it

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yes, this is what I have in mind. However, that's not a concern of this class, I figure we could have some sort of a builder for the TrackedFile's column file list, where we take care of that.

Preconditions.checkArgument(
Sets.newHashSet(newFieldIds).size() == newFieldIds.size(),
"Invalid field IDs: duplicated IDs found in: %s",
newFieldIds);
this.fieldIds = newFieldIds;
return this;
}

Builder fileFormat(FileFormat newFileFormat) {
Preconditions.checkArgument(newFileFormat != null, "Invalid file format: null");
this.fileFormat = newFileFormat;
return this;
}

Builder fileSizeInBytes(long newFileSizeInBytes) {
Preconditions.checkArgument(
newFileSizeInBytes >= 0,
"Invalid file size in bytes: %s (must be >= 0)",
newFileSizeInBytes);
this.fileSizeInBytes = newFileSizeInBytes;
return this;
}

Builder keyMetadata(ByteBuffer newKeyMetadata) {
this.keyMetadata = newKeyMetadata;
return this;
}

ColumnFile build() {
Preconditions.checkArgument(location != null, "Missing required value: location");
Preconditions.checkArgument(fieldIds != null, "Missing required value: field IDs");
Preconditions.checkArgument(fileFormat != null, "Missing required value: file format");
Preconditions.checkArgument(
fileSizeInBytes != null, "Missing required value: file size in bytes");
return new ColumnFileStruct(location, fieldIds, fileFormat, fileSizeInBytes, keyMetadata);
}
}
}
9 changes: 8 additions & 1 deletion core/src/main/java/org/apache/iceberg/TrackedFile.java
Original file line number Diff line number Diff line change
Expand Up @@ -96,6 +96,9 @@ interface TrackedFile {
"equality_ids",
Types.ListType.ofRequired(136, Types.IntegerType.get()),
"Field ids used to determine row equality in equality delete files");
Types.NestedField COLUMN_FILES =
Types.NestedField.optional(
158, "column_files", Types.ListType.ofRequired(159, ColumnFile.schema()), "Column files");

private static List<Types.NestedField> fields(
Types.StructType partitionType, Types.StructType contentStatsType) {
Expand All @@ -120,7 +123,8 @@ PARTITION_ID, PARTITION_NAME, typeOrUnknown(partitionType), PARTITION_DOC),
MANIFEST_INFO,
KEY_METADATA,
SPLIT_OFFSETS,
EQUALITY_IDS);
EQUALITY_IDS,
COLUMN_FILES);
}

private static Type typeOrUnknown(Types.StructType structType) {
Expand Down Expand Up @@ -195,6 +199,9 @@ static Schema readSchema(Types.StructType partitionType, Types.StructType conten
/** Returns the set of field IDs used for equality comparison in equality delete files. */
List<Integer> equalityIds();

/** Returns the column files for this file. */
List<ColumnFile> columnFiles();

/** Copies this tracked file. */
TrackedFile copy();

Expand Down
Loading
Loading