Documentation
¶
Index ¶
- Constants
- Variables
- func CloneDefaultValue(value any) any
- func EntriesWithProjection(fs iceio.IO, m ManifestFile, discardDeleted bool, ...) iter.Seq2[ManifestEntry, error]
- func EnvironmentContext() map[string]string
- func ExpressionEvaluator(s *Schema, unbound BooleanExpression, caseSensitive bool) (func(StructLike) (bool, error), error)
- func ExtractFieldIDs(expr BooleanExpression) ([]int, error)
- func GeneratePartitionFieldName(schema *Schema, field PartitionField) (string, error)
- func IndexByID(schema *Schema) (map[int]NestedField, error)
- func IndexByName(schema *Schema) (map[string]int, error)
- func IndexNameByID(schema *Schema) (map[int]string, error)
- func IndexParents(schema *Schema) (map[int]int, error)
- func IsMetadataColumn(fieldID int) bool
- func IsReservedFieldID(fieldID int) bool
- func NormalizeVariantPath(fields []string) string
- func PreOrderVisit[T any](sc *Schema, visitor PreOrderSchemaVisitor[T]) (res T, err error)
- func PropUInt(p Properties, key string, defVal uint) uint
- func RemoveEnvironmentProperty(key string)
- func SetEnvironmentProperty(key, value string)
- func SetSchemaCacheSize(size int) error
- func TranslateColumnNamesForScan(expr BooleanExpression, fileSchema *Schema) (BooleanExpression, []VariantExtractColumn, error)
- func Version() string
- func Visit[T any](sc *Schema, visitor SchemaVisitor[T]) (res T, err error)
- func VisitBoundPredicate[T any](e BoundPredicate, visitor BoundBooleanExprVisitor[T]) T
- func VisitBoundPredicateRef[T any](e BoundPredicate, visitor BoundBooleanExprVisitor[T], ...) T
- func VisitExpr[T any](expr BooleanExpression, visitor BooleanExprVisitor[T]) (res T, err error)
- func VisitExprEvaluator(expr BooleanExpression, visitor BooleanExprVisitor[bool]) (res bool, err error)
- func VisitMappedFields[S, T any](fields []MappedField, visitor NameMappingVisitor[S, T]) (res S, err error)
- func VisitNameMapping[S, T any](obj NameMapping, visitor NameMappingVisitor[S, T]) (res S, err error)
- func VisitSchemaWithPartner[T, P any](sc *Schema, partner P, visitor SchemaWithPartnerVisitor[T, P], ...) (res T, err error)
- func WriteManifestList(version int, out io.Writer, snapshotID int64, ...) (err error)
- type AboveMaxLiteral
- type AfterFieldVisitor
- type AfterListElementVisitor
- type AfterMapKeyVisitor
- type AfterMapValueVisitor
- type AlwaysFalse
- type AlwaysTrue
- type AndExpr
- type AvroEntryMarshaler
- type BeforeFieldVisitor
- type BeforeListElementVisitor
- type BeforeMapKeyVisitor
- type BeforeMapValueVisitor
- type BelowMinLiteral
- type BinaryLiteral
- func (b BinaryLiteral) Any() any
- func (BinaryLiteral) Comparator() Comparator[[]byte]
- func (b BinaryLiteral) Equals(other Literal) bool
- func (b BinaryLiteral) MarshalBinary() (data []byte, err error)
- func (l BinaryLiteral) MarshalJSON() ([]byte, error)
- func (b BinaryLiteral) String() string
- func (b BinaryLiteral) To(typ Type) (Literal, error)
- func (b BinaryLiteral) Type() Type
- func (b *BinaryLiteral) UnmarshalBinary(data []byte) error
- func (b BinaryLiteral) Value() []byte
- type BinaryType
- type BoolLiteral
- func (b BoolLiteral) Any() any
- func (BoolLiteral) Comparator() Comparator[bool]
- func (b BoolLiteral) Equals(l Literal) bool
- func (b BoolLiteral) MarshalBinary() (data []byte, err error)
- func (l BoolLiteral) MarshalJSON() ([]byte, error)
- func (b BoolLiteral) String() string
- func (b BoolLiteral) To(t Type) (Literal, error)
- func (b BoolLiteral) Type() Type
- func (b *BoolLiteral) UnmarshalBinary(data []byte) error
- func (b BoolLiteral) Value() bool
- type BooleanExprVisitor
- type BooleanExpression
- func BindExpr(s *Schema, expr BooleanExpression, caseSensitive bool) (BooleanExpression, error)
- func IsIn[T LiteralType](t UnboundTerm, vals ...T) BooleanExpression
- func NewAnd(left, right BooleanExpression, addl ...BooleanExpression) BooleanExpression
- func NewNot(child BooleanExpression) BooleanExpression
- func NewOr(left, right BooleanExpression, addl ...BooleanExpression) BooleanExpression
- func NotIn[T LiteralType](t UnboundTerm, vals ...T) BooleanExpression
- func ParseExpr(data []byte, schema *Schema) (BooleanExpression, error)
- func RewriteNotExpr(expr BooleanExpression) (BooleanExpression, error)
- func SanitizeExpression(expr BooleanExpression) (BooleanExpression, error)
- func SetPredicate(op Operation, t UnboundTerm, lits []Literal) BooleanExpression
- func TranslateColumnNames(expr BooleanExpression, fileSchema *Schema) (BooleanExpression, error)
- type BooleanType
- type BoundBBoxPredicate
- type BoundBooleanExprVisitor
- type BoundExtract
- type BoundGeospatialExprVisitor
- type BoundLiteralPredicate
- type BoundPredicate
- type BoundReference
- type BoundSetPredicate
- type BoundTerm
- type BoundTransform
- type BoundUnaryPredicate
- type BoundingBox
- type BucketTransform
- func (t BucketTransform) Apply(value Optional[Literal]) Optional[Literal]
- func (BucketTransform) CanTransform(t Type) bool
- func (t BucketTransform) Equals(other Transform) bool
- func (t BucketTransform) MarshalText() ([]byte, error)
- func (BucketTransform) PreservesOrder() bool
- func (t BucketTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (BucketTransform) ResultType(Type) Type
- func (t BucketTransform) String() string
- func (BucketTransform) ToHumanStr(val any) string
- func (t BucketTransform) ToHumanStrType(_ Type, val any) string
- func (t BucketTransform) Transformer(src Type) func(any) Optional[int32]
- type Comparator
- type DataFile
- type DataFileBuilder
- func (b *DataFileBuilder) BlockSizeInBytes(size int64) *DataFileBuilder
- func (b *DataFileBuilder) Build() DataFile
- func (b *DataFileBuilder) ColumnSizes(sizes map[int]int64) *DataFileBuilder
- func (b *DataFileBuilder) ContentOffset(offset int64) *DataFileBuilder
- func (b *DataFileBuilder) ContentSizeInBytes(size int64) *DataFileBuilder
- func (b *DataFileBuilder) DistinctValueCounts(counts map[int]int64) *DataFileBuilderdeprecated
- func (b *DataFileBuilder) EqualityFieldIDs(ids []int) *DataFileBuilder
- func (b *DataFileBuilder) FirstRowID(id int64) *DataFileBuilder
- func (b *DataFileBuilder) KeyMetadata(key []byte) *DataFileBuilder
- func (b *DataFileBuilder) LowerBoundValues(bounds map[int][]byte) *DataFileBuilder
- func (b *DataFileBuilder) NaNValueCounts(counts map[int]int64) *DataFileBuilder
- func (b *DataFileBuilder) NullValueCounts(counts map[int]int64) *DataFileBuilder
- func (b *DataFileBuilder) ReferencedDataFile(path string) *DataFileBuilder
- func (b *DataFileBuilder) SortOrderID(id int) *DataFileBuilder
- func (b *DataFileBuilder) SplitOffsets(offsets []int64) *DataFileBuilder
- func (b *DataFileBuilder) UpperBoundValues(bounds map[int][]byte) *DataFileBuilder
- func (b *DataFileBuilder) ValueCounts(counts map[int]int64) *DataFileBuilder
- type Date
- type DateLiteral
- func (d DateLiteral) Any() any
- func (DateLiteral) Comparator() Comparator[Date]
- func (d DateLiteral) Decrement() Literal
- func (d DateLiteral) Equals(other Literal) bool
- func (d DateLiteral) Increment() Literal
- func (d DateLiteral) MarshalBinary() (data []byte, err error)
- func (l DateLiteral) MarshalJSON() ([]byte, error)
- func (d DateLiteral) String() string
- func (d DateLiteral) To(t Type) (Literal, error)
- func (d DateLiteral) Type() Type
- func (d *DateLiteral) UnmarshalBinary(data []byte) error
- func (d DateLiteral) Value() Date
- type DateType
- type DayTransform
- func (DayTransform) Apply(value Optional[Literal]) (out Optional[Literal])
- func (t DayTransform) CanTransform(sourceType Type) bool
- func (DayTransform) Equals(other Transform) bool
- func (t DayTransform) MarshalText() ([]byte, error)
- func (DayTransform) PreservesOrder() bool
- func (t DayTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (DayTransform) ResultType(Type) Type
- func (DayTransform) String() string
- func (DayTransform) ToHumanStr(val any) string
- func (t DayTransform) ToHumanStrType(_ Type, val any) string
- func (DayTransform) Transformer(src Type) (func(any) Optional[int32], error)
- type Decimal
- type DecimalLiteral
- func (d DecimalLiteral) Any() any
- func (DecimalLiteral) Comparator() Comparator[Decimal]
- func (d DecimalLiteral) Decrement() Literal
- func (d DecimalLiteral) Equals(other Literal) bool
- func (d DecimalLiteral) Increment() Literal
- func (d DecimalLiteral) MarshalBinary() (data []byte, err error)
- func (l DecimalLiteral) MarshalJSON() ([]byte, error)
- func (d DecimalLiteral) String() string
- func (d DecimalLiteral) To(t Type) (Literal, error)
- func (d DecimalLiteral) Type() Type
- func (d *DecimalLiteral) UnmarshalBinary(data []byte) error
- func (d DecimalLiteral) Value() Decimal
- type DecimalType
- type FieldLabel
- type FieldSummary
- type FileFormat
- type FixedLiteral
- func (f FixedLiteral) Any() any
- func (FixedLiteral) Comparator() Comparator[[]byte]
- func (f FixedLiteral) Equals(other Literal) bool
- func (f FixedLiteral) MarshalBinary() (data []byte, err error)
- func (l FixedLiteral) MarshalJSON() ([]byte, error)
- func (f FixedLiteral) String() string
- func (f FixedLiteral) To(typ Type) (Literal, error)
- func (f FixedLiteral) Type() Type
- func (f *FixedLiteral) UnmarshalBinary(data []byte) error
- func (f FixedLiteral) Value() []byte
- type FixedType
- type Float32Literal
- func (f Float32Literal) Any() any
- func (Float32Literal) Comparator() Comparator[float32]
- func (f Float32Literal) Equals(other Literal) bool
- func (f Float32Literal) MarshalBinary() (data []byte, err error)
- func (l Float32Literal) MarshalJSON() ([]byte, error)
- func (f Float32Literal) String() string
- func (f Float32Literal) To(t Type) (Literal, error)
- func (f Float32Literal) Type() Type
- func (f *Float32Literal) UnmarshalBinary(data []byte) error
- func (f Float32Literal) Value() float32
- type Float32Type
- type Float64Literal
- func (f Float64Literal) Any() any
- func (Float64Literal) Comparator() Comparator[float64]
- func (f Float64Literal) Equals(other Literal) bool
- func (f Float64Literal) MarshalBinary() (data []byte, err error)
- func (l Float64Literal) MarshalJSON() ([]byte, error)
- func (f Float64Literal) String() string
- func (f Float64Literal) To(t Type) (Literal, error)
- func (f Float64Literal) Type() Type
- func (f *Float64Literal) UnmarshalBinary(data []byte) error
- func (f Float64Literal) Value() float64
- type Float64Type
- type GeoLiteral
- type GeographyType
- type GeometryType
- type HourTransform
- func (HourTransform) Apply(value Optional[Literal]) (out Optional[Literal])
- func (t HourTransform) CanTransform(sourceType Type) bool
- func (HourTransform) Equals(other Transform) bool
- func (t HourTransform) MarshalText() ([]byte, error)
- func (HourTransform) PreservesOrder() bool
- func (t HourTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (HourTransform) ResultType(Type) Type
- func (HourTransform) String() string
- func (HourTransform) ToHumanStr(val any) string
- func (t HourTransform) ToHumanStrType(_ Type, val any) string
- func (HourTransform) Transformer(src Type) (func(any) Optional[int32], error)
- type IdentityTransform
- func (IdentityTransform) Apply(value Optional[Literal]) Optional[Literal]
- func (IdentityTransform) CanTransform(t Type) bool
- func (IdentityTransform) Equals(other Transform) bool
- func (t IdentityTransform) MarshalText() ([]byte, error)
- func (IdentityTransform) PreservesOrder() bool
- func (t IdentityTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (IdentityTransform) ResultType(t Type) Type
- func (IdentityTransform) String() string
- func (IdentityTransform) ToHumanStr(val any) string
- func (t IdentityTransform) ToHumanStrType(typ Type, val any) string
- type Int32Literal
- func (i Int32Literal) Any() any
- func (Int32Literal) Comparator() Comparator[int32]
- func (i Int32Literal) Decrement() Literal
- func (i Int32Literal) Equals(other Literal) bool
- func (i Int32Literal) Increment() Literal
- func (i Int32Literal) MarshalBinary() (data []byte, err error)
- func (l Int32Literal) MarshalJSON() ([]byte, error)
- func (i Int32Literal) String() string
- func (i Int32Literal) To(t Type) (Literal, error)
- func (i Int32Literal) Type() Type
- func (i *Int32Literal) UnmarshalBinary(data []byte) error
- func (i Int32Literal) Value() int32
- type Int32Type
- type Int64Literal
- func (i Int64Literal) Any() any
- func (Int64Literal) Comparator() Comparator[int64]
- func (i Int64Literal) Decrement() Literal
- func (i Int64Literal) Equals(other Literal) bool
- func (i Int64Literal) Increment() Literal
- func (i Int64Literal) MarshalBinary() (data []byte, err error)
- func (l Int64Literal) MarshalJSON() ([]byte, error)
- func (i Int64Literal) String() string
- func (i Int64Literal) To(t Type) (Literal, error)
- func (i Int64Literal) Type() Type
- func (i *Int64Literal) UnmarshalBinary(data []byte) error
- func (i Int64Literal) Value() int64
- type Int64Type
- type Labels
- type ListType
- type Literal
- func CastVariantLiteral(v variant.Value, typ PrimitiveType) (Literal, bool)
- func Float32AboveMaxLiteral() Literal
- func Float32BelowMinLiteral() Literal
- func Float64AboveMaxLiteral() Literal
- func Float64BelowMinLiteral() Literal
- func Int32AboveMaxLiteral() Literal
- func Int32BelowMinLiteral() Literal
- func Int64AboveMaxLiteral() Literal
- func Int64BelowMinLiteral() Literal
- func LiteralFromBytes(typ Type, data []byte) (Literal, error)
- func NewLiteral[T LiteralType](val T) Literal
- type LiteralType
- type ManifestBuilder
- func (b *ManifestBuilder) AddedFiles(cnt int32) *ManifestBuilder
- func (b *ManifestBuilder) AddedRows(cnt int64) *ManifestBuilder
- func (b *ManifestBuilder) Build() ManifestFile
- func (b *ManifestBuilder) Content(content ManifestContent) *ManifestBuilder
- func (b *ManifestBuilder) DeletedFiles(cnt int32) *ManifestBuilder
- func (b *ManifestBuilder) DeletedRows(cnt int64) *ManifestBuilder
- func (b *ManifestBuilder) ExistingFiles(cnt int32) *ManifestBuilder
- func (b *ManifestBuilder) ExistingRows(cnt int64) *ManifestBuilder
- func (b *ManifestBuilder) KeyMetadata(km []byte) *ManifestBuilder
- func (b *ManifestBuilder) Partitions(p []FieldSummary) *ManifestBuilder
- func (b *ManifestBuilder) SequenceNum(num, minSeqNum int64) *ManifestBuilder
- type ManifestContent
- type ManifestEntry
- type ManifestEntryBuilder
- type ManifestEntryContent
- type ManifestEntryProjection
- type ManifestEntryStatus
- type ManifestFile
- func ReadManifestList(in io.Reader) ([]ManifestFile, error)
- func WriteManifest(filename string, out io.Writer, version int, spec PartitionSpec, ...) (mf ManifestFile, err error)
- func WriteManifestV3(filename string, out io.Writer, firstRowID int64, spec PartitionSpec, ...) (mf ManifestFile, nextFirstRowID int64, err error)
- type ManifestFileOption
- type ManifestListWriter
- func NewManifestListWriterV1(out io.Writer, snapshotID int64, parentSnapshot *int64) (*ManifestListWriter, error)
- func NewManifestListWriterV2(out io.Writer, snapshotID, sequenceNumber int64, parentSnapshot *int64) (*ManifestListWriter, error)
- func NewManifestListWriterV3(out io.Writer, snapshotId, sequenceNumber, firstRowID int64, ...) (*ManifestListWriter, error)
- type ManifestReader
- func (c *ManifestReader) Close() error
- func (c *ManifestReader) ManifestContent() ManifestContent
- func (c *ManifestReader) PartitionSpec() (*PartitionSpec, error)
- func (c *ManifestReader) PartitionSpecID() (int, error)
- func (c *ManifestReader) ReadEntry() (ManifestEntry, error)
- func (c *ManifestReader) Schema() (*Schema, error)
- func (c *ManifestReader) SchemaID() (int, error)
- func (c *ManifestReader) Version() int
- type ManifestWriter
- func (w *ManifestWriter) Add(entry ManifestEntry) error
- func (w *ManifestWriter) Close() error
- func (w *ManifestWriter) Delete(entry ManifestEntry) error
- func (w *ManifestWriter) Existing(entry ManifestEntry) error
- func (w *ManifestWriter) ToManifestFile(location string, length int64, opts ...ManifestFileOption) (ManifestFile, error)
- type ManifestWriterOption
- type MapType
- func (m *MapType) Equals(other Type) bool
- func (m *MapType) Fields() []NestedField
- func (m *MapType) KeyField() NestedField
- func (m *MapType) MarshalJSON() ([]byte, error)
- func (m *MapType) String() string
- func (*MapType) Type() string
- func (m *MapType) UnmarshalJSON(b []byte) error
- func (m *MapType) ValueField() NestedField
- type MappedField
- type MonthTransform
- func (MonthTransform) Apply(value Optional[Literal]) (out Optional[Literal])
- func (t MonthTransform) CanTransform(sourceType Type) bool
- func (MonthTransform) Equals(other Transform) bool
- func (t MonthTransform) MarshalText() ([]byte, error)
- func (MonthTransform) PreservesOrder() bool
- func (t MonthTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (MonthTransform) ResultType(Type) Type
- func (MonthTransform) String() string
- func (t MonthTransform) ToHumanStr(val any) string
- func (t MonthTransform) ToHumanStrType(_ Type, val any) string
- func (MonthTransform) Transformer(src Type) (func(any) Optional[int32], error)
- type NameMapping
- type NameMappingAccessor
- func (NameMappingAccessor) FieldPartner(partnerStruct *MappedField, _ int, fieldName string) *MappedField
- func (NameMappingAccessor) ListElementPartner(partnerList *MappedField) *MappedField
- func (NameMappingAccessor) MapKeyPartner(partnerMap *MappedField) *MappedField
- func (NameMappingAccessor) MapValuePartner(partnerMap *MappedField) *MappedField
- func (NameMappingAccessor) SchemaPartner(partner *MappedField) *MappedField
- type NameMappingVisitor
- type NestedField
- type NestedType
- type NotExpr
- type NumericLiteral
- type Operation
- type Optional
- type OrExpr
- type PartitionField
- type PartitionOption
- type PartitionSpec
- func (p *PartitionSpec) BindToSchema(schema *Schema, lastPartitionID *int, newSpecID *int) (PartitionSpec, error)
- func (ps *PartitionSpec) Clone() PartitionSpec
- func (ps *PartitionSpec) CompatibleWith(other *PartitionSpec) bool
- func (ps PartitionSpec) Equals(other PartitionSpec) bool
- func (ps *PartitionSpec) Field(i int) PartitionField
- func (ps *PartitionSpec) Fields() iter.Seq2[int, PartitionField]
- func (ps *PartitionSpec) FieldsBySourceID(fieldID int) []PartitionField
- func (ps *PartitionSpec) FieldsRef(_ internal.PartitionSpecRef) []PartitionField
- func (ps *PartitionSpec) ID() int
- func (ps PartitionSpec) IsUnpartitioned() bool
- func (ps *PartitionSpec) LastAssignedFieldID() int
- func (p *PartitionSpec) Len() int
- func (ps PartitionSpec) MarshalJSON() ([]byte, error)
- func (ps *PartitionSpec) NumFields() int
- func (ps *PartitionSpec) PartitionToPath(data StructLike, sc *Schema) string
- func (ps *PartitionSpec) PartitionType(schema *Schema) *StructType
- func (ps PartitionSpec) String() string
- func (ps *PartitionSpec) UnmarshalJSON(b []byte) error
- type PartnerAccessor
- type PreOrderSchemaVisitor
- type PrimitiveType
- type Properties
- type Reference
- type Schema
- func ApplyNameMapping(schemaWithoutIDs *Schema, nameMapping NameMapping) (*Schema, error)
- func AssignFreshSchemaIDs(sc *Schema, nextID func() int) (*Schema, error)
- func NewSchema(id int, fields ...NestedField) *Schema
- func NewSchemaFromJsonFields(id int, jsonFieldsStr string) (*Schema, error)
- func NewSchemaWithIdentifiers(id int, identifierIDs []int, fields ...NestedField) *Schema
- func PruneColumns(schema *Schema, selected map[int]Void, selectFullTypes bool) (*Schema, error)
- func SanitizeColumnNames(sc *Schema) (*Schema, error)
- func SchemaWithRowID(s *Schema) *Schema
- func SchemaWithRowLineage(s *Schema) *Schema
- func SchemaWithRowLineageColumns(s *Schema, rowID, lastUpdatedSeq bool) *Schema
- func (s *Schema) AsStruct() StructType
- func (s *Schema) Equals(other *Schema) bool
- func (s *Schema) Field(i int) NestedField
- func (s *Schema) FieldHasOptionalParent(id int) bool
- func (s *Schema) FieldIDs() []int
- func (s *Schema) Fields() []NestedField
- func (s *Schema) FieldsRef(_ internal.SchemaRef) []NestedField
- func (s *Schema) FindColumnName(fieldID int) (string, bool)
- func (s *Schema) FindFieldByID(id int) (NestedField, bool)
- func (s *Schema) FindFieldByIDRef(id int, _ internal.SchemaRef) (NestedField, bool)
- func (s *Schema) FindFieldByName(name string) (NestedField, bool)
- func (s *Schema) FindFieldByNameCaseInsensitive(name string) (NestedField, bool)
- func (s *Schema) FindTypeByID(id int) (Type, bool)
- func (s *Schema) FindTypeByName(name string) (Type, bool)
- func (s *Schema) FindTypeByNameCaseInsensitive(name string) (Type, bool)
- func (s *Schema) FlatFields() (iter.Seq[NestedField], error)
- func (s *Schema) HighestFieldID() int
- func (s *Schema) MarshalJSON() ([]byte, error)
- func (s *Schema) NameMapping() NameMapping
- func (s *Schema) NumFields() int
- func (s *Schema) Select(caseSensitive bool, names ...string) (*Schema, error)
- func (s *Schema) String() string
- func (s *Schema) Type() string
- func (s *Schema) UnmarshalJSON(b []byte) error
- type SchemaVisitor
- type SchemaVisitorPerPrimitiveType
- type SchemaWithPartnerVisitor
- type Set
- type StringLiteral
- func (s StringLiteral) Any() any
- func (StringLiteral) Comparator() Comparator[string]
- func (s StringLiteral) Equals(other Literal) bool
- func (s StringLiteral) MarshalBinary() (data []byte, err error)
- func (l StringLiteral) MarshalJSON() ([]byte, error)
- func (s StringLiteral) String() string
- func (s StringLiteral) To(typ Type) (Literal, error)
- func (s StringLiteral) Type() Type
- func (s *StringLiteral) UnmarshalBinary(data []byte) error
- func (s StringLiteral) Value() string
- type StringType
- type StructLike
- type StructType
- type Term
- type Time
- type TimeLiteral
- func (t TimeLiteral) Any() any
- func (TimeLiteral) Comparator() Comparator[Time]
- func (t TimeLiteral) Equals(other Literal) bool
- func (t TimeLiteral) MarshalBinary() (data []byte, err error)
- func (l TimeLiteral) MarshalJSON() ([]byte, error)
- func (t TimeLiteral) String() string
- func (t TimeLiteral) To(typ Type) (Literal, error)
- func (t TimeLiteral) Type() Type
- func (t *TimeLiteral) UnmarshalBinary(data []byte) error
- func (t TimeLiteral) Value() Time
- type TimeTransform
- type TimeType
- type Timestamp
- type TimestampLiteral
- func (t TimestampLiteral) Any() any
- func (TimestampLiteral) Comparator() Comparator[Timestamp]
- func (t TimestampLiteral) Decrement() Literal
- func (t TimestampLiteral) Equals(other Literal) bool
- func (t TimestampLiteral) Increment() Literal
- func (t TimestampLiteral) MarshalBinary() (data []byte, err error)
- func (TimestampLiteral) MarshalJSON() ([]byte, error)
- func (t TimestampLiteral) String() string
- func (t TimestampLiteral) To(typ Type) (Literal, error)
- func (t TimestampLiteral) Type() Type
- func (t *TimestampLiteral) UnmarshalBinary(data []byte) error
- func (t TimestampLiteral) Value() Timestamp
- type TimestampNano
- type TimestampNsLiteral
- func (t TimestampNsLiteral) Any() any
- func (TimestampNsLiteral) Comparator() Comparator[TimestampNano]
- func (t TimestampNsLiteral) Decrement() Literal
- func (t TimestampNsLiteral) Equals(other Literal) bool
- func (t TimestampNsLiteral) Increment() Literal
- func (t TimestampNsLiteral) MarshalBinary() (data []byte, err error)
- func (TimestampNsLiteral) MarshalJSON() ([]byte, error)
- func (t TimestampNsLiteral) String() string
- func (t TimestampNsLiteral) To(typ Type) (Literal, error)
- func (t TimestampNsLiteral) Type() Type
- func (t *TimestampNsLiteral) UnmarshalBinary(data []byte) error
- func (t TimestampNsLiteral) Value() TimestampNano
- type TimestampNsType
- type TimestampType
- type TimestampTzNsType
- type TimestampTzType
- type Transform
- type TruncateTransform
- func (t TruncateTransform) Apply(value Optional[Literal]) (out Optional[Literal])
- func (TruncateTransform) CanTransform(t Type) bool
- func (t TruncateTransform) Equals(other Transform) bool
- func (t TruncateTransform) MarshalText() ([]byte, error)
- func (TruncateTransform) PreservesOrder() bool
- func (t TruncateTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (TruncateTransform) ResultType(t Type) Type
- func (t TruncateTransform) String() string
- func (TruncateTransform) ToHumanStr(val any) string
- func (t TruncateTransform) ToHumanStrType(_ Type, val any) string
- func (t TruncateTransform) Transformer(src Type) (func(any) any, error)
- type Type
- type TypedLiteral
- type UUIDLiteral
- func (u UUIDLiteral) Any() any
- func (UUIDLiteral) Comparator() Comparator[uuid.UUID]
- func (u UUIDLiteral) Equals(other Literal) bool
- func (u UUIDLiteral) MarshalBinary() (data []byte, err error)
- func (l UUIDLiteral) MarshalJSON() ([]byte, error)
- func (u UUIDLiteral) String() string
- func (u UUIDLiteral) To(typ Type) (Literal, error)
- func (UUIDLiteral) Type() Type
- func (u *UUIDLiteral) UnmarshalBinary(data []byte) error
- func (u UUIDLiteral) Value() uuid.UUID
- type UUIDType
- type UnboundPartitionSpec
- type UnboundPredicate
- func BBoxIntersects(t UnboundTerm, bbox BoundingBox) UnboundPredicate
- func EqualTo[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func GreaterThan[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func GreaterThanEqual[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func IsNaN(t UnboundTerm) UnboundPredicate
- func IsNull(t UnboundTerm) UnboundPredicate
- func LessThan[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func LessThanEqual[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func LiteralPredicate(op Operation, t UnboundTerm, lit Literal) UnboundPredicate
- func NotEqualTo[T LiteralType](t UnboundTerm, v T) UnboundPredicate
- func NotNaN(t UnboundTerm) UnboundPredicate
- func NotNull(t UnboundTerm) UnboundPredicate
- func NotStartsWith(t UnboundTerm, v string) UnboundPredicate
- func StartsWith(t UnboundTerm, v string) UnboundPredicate
- func UnaryPredicate(op Operation, t UnboundTerm) UnboundPredicate
- type UnboundTerm
- type UnboundTransform
- type UnknownTransform
- func (UnknownTransform) Apply(Optional[Literal]) Optional[Literal]
- func (UnknownTransform) CanTransform(Type) bool
- func (t UnknownTransform) Equals(other Transform) bool
- func (t UnknownTransform) MarshalText() ([]byte, error)
- func (UnknownTransform) PreservesOrder() bool
- func (UnknownTransform) Project(string, BoundPredicate) (UnboundPredicate, error)
- func (UnknownTransform) ResultType(Type) Type
- func (t UnknownTransform) String() string
- func (UnknownTransform) ToHumanStr(val any) string
- func (UnknownTransform) ToHumanStrType(typ Type, val any) string
- type UnknownType
- type VariantExtractColumn
- type VariantLiteral
- func (v VariantLiteral) Any() any
- func (VariantLiteral) Comparator() Comparator[variant.Value]
- func (v VariantLiteral) Equals(other Literal) bool
- func (v VariantLiteral) MarshalBinary() ([]byte, error)
- func (v VariantLiteral) String() string
- func (v VariantLiteral) To(typ Type) (Literal, error)
- func (VariantLiteral) Type() Type
- func (v VariantLiteral) Value() variant.Value
- type VariantType
- type Void
- type VoidTransform
- func (VoidTransform) Apply(value Optional[Literal]) Optional[Literal]
- func (VoidTransform) CanTransform(Type) bool
- func (VoidTransform) Equals(other Transform) bool
- func (t VoidTransform) MarshalText() ([]byte, error)
- func (VoidTransform) PreservesOrder() bool
- func (VoidTransform) Project(string, BoundPredicate) (UnboundPredicate, error)
- func (VoidTransform) ResultType(t Type) Type
- func (VoidTransform) String() string
- func (VoidTransform) ToHumanStr(any) string
- func (VoidTransform) ToHumanStrType(Type, any) string
- type YearTransform
- func (YearTransform) Apply(value Optional[Literal]) (out Optional[Literal])
- func (t YearTransform) CanTransform(sourceType Type) bool
- func (YearTransform) Equals(other Transform) bool
- func (t YearTransform) MarshalText() ([]byte, error)
- func (YearTransform) PreservesOrder() bool
- func (t YearTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
- func (YearTransform) ResultType(Type) Type
- func (YearTransform) String() string
- func (YearTransform) ToHumanStr(val any) string
- func (t YearTransform) ToHumanStrType(_ Type, val any) string
- func (YearTransform) Transformer(src Type) (func(any) Optional[int32], error)
Constants ¶
const ( // EnvironmentEngineNameKey identifies the engine name property. EnvironmentEngineNameKey = "engine-name" // EnvironmentEngineVersionKey identifies the engine version property. EnvironmentEngineVersionKey = "engine-version" )
const ( // RowIDFieldID is the field ID for _row_id (optional long). A unique long identifier for every row. // Reserved as Integer.MAX_VALUE - 107. RowIDFieldID = 2147483540 // LastUpdatedSequenceNumberFieldID is the field ID for _last_updated_sequence_number (optional long). // The sequence number of the commit that last updated the row. Reserved as Integer.MAX_VALUE - 108. LastUpdatedSequenceNumberFieldID = 2147483539 )
Row lineage metadata column field IDs (v3+). Reserved IDs are Integer.MAX_VALUE - 107 and Integer.MAX_VALUE - 108 per the Iceberg spec (Metadata Columns / Row Lineage).
const ( RowIDColumnName = "_row_id" LastUpdatedSequenceNumberColumnName = "_last_updated_sequence_number" )
Row lineage metadata column names (v3+).
const ( PartitionDataIDStart = 1000 InitialPartitionSpecID = 0 )
const DefaultGeoCRS = "OGC:CRS84"
DefaultGeoCRS is the CRS of the geometry and geography types when no CRS is given; a Parquet GEOMETRY or GEOGRAPHY logical type without a CRS means the same value.
const MaxStructFieldID = math.MaxInt32 - 200
MaxStructFieldID is the largest field ID a user-supplied schema may assign. Field IDs greater than this are reserved by the spec for metadata columns (_row_id, _last_updated_sequence_number, _file, _pos, _deleted, _spec_id, _partition, ...). Java: TypeUtil.MAX_STRUCT_FIELD_ID (Integer.MAX_VALUE - 200).
Variables ¶
var ( ErrInvalidTypeString = errors.New("invalid type") ErrNotImplemented = errors.New("not implemented") ErrInvalidArgument = errors.New("invalid argument") ErrInvalidFormatVersion = fmt.Errorf("%w: invalid format version", ErrInvalidArgument) ErrInvalidSchema = errors.New("invalid schema") ErrInvalidPartitionSpec = errors.New("invalid partition spec") ErrEmptyManifest = errors.New("empty manifest file has been written") ErrInvalidTransform = errors.New("invalid transform syntax") ErrType = errors.New("type error") ErrBadCast = errors.New("could not cast value") ErrBadLiteral = errors.New("invalid literal value") ErrInvalidBinSerialization = errors.New("invalid binary serialization") ErrInvalidFixedLength = errors.New("invalid fixed literal length") ErrResolve = errors.New("cannot resolve type") // ErrBBoxNotSerializable is returned when marshaling a geospatial bbox // predicate to REST expression JSON: bbox predicates exist only for local // scan planning and have no representation in the REST expression grammar. // It wraps ErrNotImplemented (not ErrInvalidArgument) because this is a // contract/not-implemented condition, not a bad caller argument - matching // how the substrait backstop surfaces the same predicate. ErrBBoxNotSerializable = fmt.Errorf("%w: geospatial bbox predicates cannot be serialized to REST expression JSON", ErrNotImplemented) // ErrExtractNotSerializable is returned when marshaling a variant extract term // to REST expression JSON: extract terms have no representation in the REST // expression grammar and would otherwise encode to an empty object. ErrExtractNotSerializable = fmt.Errorf("%w: variant extract terms cannot be serialized to REST expression JSON", ErrNotImplemented) )
var PositionalDeleteSchema = NewSchema(0, NestedField{ID: 2147483546, Type: PrimitiveTypes.String, Name: "file_path", Required: true}, NestedField{ID: 2147483545, Type: PrimitiveTypes.Int64, Name: "pos", Required: true}, )
var PrimitiveTypes = struct { Bool PrimitiveType Int32 PrimitiveType Int64 PrimitiveType Float32 PrimitiveType Float64 PrimitiveType Date PrimitiveType Time PrimitiveType Timestamp PrimitiveType TimestampTz PrimitiveType TimestampNs PrimitiveType TimestampTzNs PrimitiveType String PrimitiveType Binary PrimitiveType UUID PrimitiveType Unknown PrimitiveType }{ Bool: BooleanType{}, Int32: Int32Type{}, Int64: Int64Type{}, Float32: Float32Type{}, Float64: Float64Type{}, Date: DateType{}, Time: TimeType{}, Timestamp: TimestampType{}, TimestampTz: TimestampTzType{}, TimestampNs: TimestampNsType{}, TimestampTzNs: TimestampTzNsType{}, String: StringType{}, Binary: BinaryType{}, UUID: UUIDType{}, Unknown: UnknownType{}, }
var UnpartitionedSpec = &PartitionSpec{id: 0}
UnpartitionedSpec is the default unpartitioned spec which can be used for comparisons or to just provide a convenience for referencing the same unpartitioned spec object.
Functions ¶
func CloneDefaultValue ¶ added in v0.7.0
CloneDefaultValue returns a deep copy of an Iceberg initial or write default value.
func EntriesWithProjection ¶ added in v0.7.0
func EntriesWithProjection( fs iceio.IO, m ManifestFile, discardDeleted bool, projection ManifestEntryProjection, ) iter.Seq2[ManifestEntry, error]
EntriesWithProjection streams manifest entries using a reader-schema projection. It is the projected counterpart to ManifestFile.Entries and is useful when a caller needs only the fields required for scan planning. The returned entries must not be passed to a ManifestWriter, because the projection may omit statistics that would be lost on rewrite. ManifestWriter and the DataFile Avro codec reject them with ErrInvalidArgument.
func EnvironmentContext ¶ added in v0.7.0
EnvironmentContext returns an independent snapshot of the process-wide context used to populate report metadata and persisted snapshot summaries. The returned map may be modified by the caller without changing the stored context.
func ExpressionEvaluator ¶
func ExpressionEvaluator(s *Schema, unbound BooleanExpression, caseSensitive bool) (func(StructLike) (bool, error), error)
ExpressionEvaluator returns a function which can be used to evaluate a given expression as long as a structlike value is passed which operates like and matches the passed in schema.
func ExtractFieldIDs ¶
func ExtractFieldIDs(expr BooleanExpression) ([]int, error)
ExtractFieldIDs returns a slice containing the field IDs which are referenced by any terms in the given expression. This enables retrieving exactly which fields are needed for an expression.
func GeneratePartitionFieldName ¶ added in v0.4.0
func GeneratePartitionFieldName(schema *Schema, field PartitionField) (string, error)
GeneratePartitionFieldName returns default partition field name based on field transform type
The default names are aligned with other client implementations https://github.com/apache/iceberg/blob/main/core/src/main/java/org/apache/iceberg/BaseUpdatePartitionSpec.java#L518-L563
func IndexByID ¶
func IndexByID(schema *Schema) (map[int]NestedField, error)
IndexByID performs a post-order traversal of the given schema and returns a mapping from field ID to field.
func IndexByName ¶
IndexByName performs a post-order traversal of the schema and returns a mapping from field name to field ID.
func IndexNameByID ¶
IndexNameByID performs a post-order traversal of the schema and returns a mapping from field ID to field name.
func IndexParents ¶
IndexParents generates an index of field IDs to their parent field IDs. Root fields are not indexed
func IsMetadataColumn ¶ added in v0.6.0
IsMetadataColumn returns true if the field ID is a reserved metadata column (e.g. row lineage).
func IsReservedFieldID ¶ added in v0.7.0
IsReservedFieldID reports whether fieldID falls in the range the spec reserves for metadata columns. User schemas must not assign IDs in this range; table schema field IDs must not exceed MaxStructFieldID.
func NormalizeVariantPath ¶ added in v0.7.0
NormalizeVariantPath renders member names as the spec's RFC-9535 normalized JSON path. Exported for table/internal; not part of the stable public API.
func PreOrderVisit ¶ added in v0.2.0
func PreOrderVisit[T any](sc *Schema, visitor PreOrderSchemaVisitor[T]) (res T, err error)
func PropUInt ¶ added in v0.6.0
func PropUInt(p Properties, key string, defVal uint) uint
PropUInt reads an unsigned-integer property by key, preserving the legacy fallback behavior while avoiding truncation on platforms where uint is narrower than uint64.
func RemoveEnvironmentProperty ¶ added in v0.7.0
func RemoveEnvironmentProperty(key string)
RemoveEnvironmentProperty removes one process-wide environment context property.
func SetEnvironmentProperty ¶ added in v0.7.0
func SetEnvironmentProperty(key, value string)
SetEnvironmentProperty sets one process-wide environment context property.
func SetSchemaCacheSize ¶ added in v0.6.0
SetSchemaCacheSize resizes the manifest-entry schema cache used by the DataFile codec. The default capacity is sized for a few thousand active partition specs; long-running consumers with larger working sets (e.g. a compaction service touching many tables) should raise it. Existing entries are preserved on grow; on shrink, least-recently used entries are evicted down to the new size. Safe to call concurrently with codec operations: the underlying golang-lru/v2 cache serializes Resize against Get/Add through the same mutex.
func TranslateColumnNamesForScan ¶ added in v0.7.0
func TranslateColumnNamesForScan(expr BooleanExpression, fileSchema *Schema) (BooleanExpression, []VariantExtractColumn, error)
TranslateColumnNamesForScan translates a bound filter to the file schema, mapping variant extract terms to synthetic reference columns.
func Visit ¶
func Visit[T any](sc *Schema, visitor SchemaVisitor[T]) (res T, err error)
Visit accepts a visitor and performs a post-order traversal of the given schema.
func VisitBoundPredicate ¶
func VisitBoundPredicate[T any](e BoundPredicate, visitor BoundBooleanExprVisitor[T]) T
VisitBoundPredicate uses a BoundBooleanExprVisitor to call the appropriate method based on the type of operation in the predicate. This is a convenience function for implementing the VisitBound method of a BoundBooleanExprVisitor by simply calling iceberg.VisitBoundPredicate(pred, this).
If the predicate is a geospatial bbox op (OpBBoxIntersects/OpBBoxNotIntersects) and the visitor does not also implement BoundGeospatialExprVisitor[T], this panics with an error wrapping ErrNotImplemented. When reached through VisitExpr that panic is recovered into a returned error; a caller invoking this directly must implement the extension or recover the panic itself. Set predicates are dispatched with detached literal sets. Trusted built-in visitors that need the zero-copy path can use VisitBoundPredicateRef.
func VisitBoundPredicateRef ¶ added in v0.7.0
func VisitBoundPredicateRef[T any](e BoundPredicate, visitor BoundBooleanExprVisitor[T], _ internal.BoundPredicateRef) T
VisitBoundPredicateRef is the zero-copy dispatch path for trusted visitors inside this module. It is gated by an internal token so external visitors use VisitBoundPredicate and receive a detached set.
func VisitExpr ¶
func VisitExpr[T any](expr BooleanExpression, visitor BooleanExprVisitor[T]) (res T, err error)
VisitExpr is a convenience function to use a given visitor to visit all parts of a boolean expression in-order. Values returned from the methods are passed to the subsequent methods, effectively "bubbling up" the results.
func VisitExprEvaluator ¶ added in v0.7.0
func VisitExprEvaluator(expr BooleanExpression, visitor BooleanExprVisitor[bool]) (res bool, err error)
VisitExprEvaluator traverses a boolean expression for a visitor whose boolean results represent expression truth. Unlike VisitExpr, it only visits the nodes required to determine the result: it skips the right side of an AND when the left side is false and the right side of an OR when the left side is true.
func VisitMappedFields ¶ added in v0.2.0
func VisitMappedFields[S, T any](fields []MappedField, visitor NameMappingVisitor[S, T]) (res S, err error)
func VisitNameMapping ¶ added in v0.2.0
func VisitNameMapping[S, T any](obj NameMapping, visitor NameMappingVisitor[S, T]) (res S, err error)
func VisitSchemaWithPartner ¶
func VisitSchemaWithPartner[T, P any](sc *Schema, partner P, visitor SchemaWithPartnerVisitor[T, P], accessor PartnerAccessor[P]) (res T, err error)
func WriteManifestList ¶
func WriteManifestList(version int, out io.Writer, snapshotID int64, parentSnapshotID, sequenceNumber *int64, firstRowId int64, files []ManifestFile) (err error)
WriteManifestList writes a list of manifest files to an avro file.
Types ¶
type AboveMaxLiteral ¶
type AboveMaxLiteral interface {
Literal
// contains filtered or unexported methods
}
AboveMaxLiteral represents values that are above the maximum for their type such as values > math.MaxInt32 for an Int32Literal
type AfterFieldVisitor ¶
type AfterFieldVisitor interface {
AfterField(field NestedField)
}
type AfterListElementVisitor ¶
type AfterListElementVisitor interface {
AfterListElement(elem NestedField)
}
type AfterMapKeyVisitor ¶
type AfterMapKeyVisitor interface {
AfterMapKey(key NestedField)
}
type AfterMapValueVisitor ¶
type AfterMapValueVisitor interface {
AfterMapValue(value NestedField)
}
type AlwaysFalse ¶
type AlwaysFalse struct{}
AlwaysFalse is the boolean expression "False"
func (AlwaysFalse) Equals ¶
func (AlwaysFalse) Equals(other BooleanExpression) bool
func (AlwaysFalse) MarshalJSON ¶ added in v0.7.0
func (AlwaysFalse) MarshalJSON() ([]byte, error)
func (AlwaysFalse) Negate ¶
func (AlwaysFalse) Negate() BooleanExpression
func (AlwaysFalse) Op ¶
func (AlwaysFalse) Op() Operation
func (AlwaysFalse) String ¶
func (AlwaysFalse) String() string
type AlwaysTrue ¶
type AlwaysTrue struct{}
AlwaysTrue is the boolean expression "True"
func (AlwaysTrue) Equals ¶
func (AlwaysTrue) Equals(other BooleanExpression) bool
func (AlwaysTrue) MarshalJSON ¶ added in v0.7.0
func (AlwaysTrue) MarshalJSON() ([]byte, error)
MarshalJSON emits the REST form, so an expression can be used as a "filter" field directly. Tag such fields omitempty to drop a nil one.
func (AlwaysTrue) Negate ¶
func (AlwaysTrue) Negate() BooleanExpression
func (AlwaysTrue) Op ¶
func (AlwaysTrue) Op() Operation
func (AlwaysTrue) String ¶
func (AlwaysTrue) String() string
type AndExpr ¶
type AndExpr struct {
// contains filtered or unexported fields
}
func (AndExpr) Equals ¶
func (a AndExpr) Equals(other BooleanExpression) bool
func (AndExpr) MarshalJSON ¶ added in v0.7.0
func (AndExpr) Negate ¶
func (a AndExpr) Negate() BooleanExpression
type AvroEntryMarshaler ¶ added in v0.6.0
type AvroEntryMarshaler interface {
MarshalAvroEntry(spec PartitionSpec, schema *Schema, version int) ([]byte, error)
}
AvroEntryMarshaler is implemented by DataFile values that can be encoded using the manifest-entry Avro encoding. The iceberg package's built-in DataFile implementation satisfies it; external implementations can also satisfy it to participate in the github.com/apache/iceberg-go/codec DataFile codec.
The encoded bytes are the same bytes a manifest carries for this data file. Implementations must produce output that the iceberg package's manifest-entry Avro decoder accepts. The built-in DataFile implementation rejects projected or stat-stripped files with ErrInvalidArgument because omitted metadata cannot be recovered.
type BeforeFieldVisitor ¶
type BeforeFieldVisitor interface {
BeforeField(field NestedField)
}
type BeforeListElementVisitor ¶
type BeforeListElementVisitor interface {
BeforeListElement(elem NestedField)
}
type BeforeMapKeyVisitor ¶
type BeforeMapKeyVisitor interface {
BeforeMapKey(key NestedField)
}
type BeforeMapValueVisitor ¶
type BeforeMapValueVisitor interface {
BeforeMapValue(value NestedField)
}
type BelowMinLiteral ¶
type BelowMinLiteral interface {
Literal
// contains filtered or unexported methods
}
BelowMinLiteral represents values that are below the minimum for their type such as values < math.MinInt32 for an Int32Literal
type BinaryLiteral ¶
type BinaryLiteral []byte
func (BinaryLiteral) Any ¶ added in v0.2.0
func (b BinaryLiteral) Any() any
func (BinaryLiteral) Comparator ¶
func (BinaryLiteral) Comparator() Comparator[[]byte]
func (BinaryLiteral) Equals ¶
func (b BinaryLiteral) Equals(other Literal) bool
func (BinaryLiteral) MarshalBinary ¶
func (b BinaryLiteral) MarshalBinary() (data []byte, err error)
func (BinaryLiteral) MarshalJSON ¶ added in v0.7.0
func (l BinaryLiteral) MarshalJSON() ([]byte, error)
func (BinaryLiteral) String ¶
func (b BinaryLiteral) String() string
func (BinaryLiteral) Type ¶
func (b BinaryLiteral) Type() Type
func (*BinaryLiteral) UnmarshalBinary ¶
func (b *BinaryLiteral) UnmarshalBinary(data []byte) error
func (BinaryLiteral) Value ¶
func (b BinaryLiteral) Value() []byte
type BinaryType ¶
type BinaryType struct{}
func (BinaryType) Equals ¶
func (BinaryType) Equals(other Type) bool
func (BinaryType) String ¶
func (BinaryType) String() string
func (BinaryType) Type ¶
func (BinaryType) Type() string
type BoolLiteral ¶
type BoolLiteral bool
func (BoolLiteral) Any ¶ added in v0.2.0
func (b BoolLiteral) Any() any
func (BoolLiteral) Comparator ¶
func (BoolLiteral) Comparator() Comparator[bool]
func (BoolLiteral) Equals ¶
func (b BoolLiteral) Equals(l Literal) bool
func (BoolLiteral) MarshalBinary ¶
func (b BoolLiteral) MarshalBinary() (data []byte, err error)
func (BoolLiteral) MarshalJSON ¶ added in v0.7.0
func (l BoolLiteral) MarshalJSON() ([]byte, error)
Each literal writes its REST form (Java's SingleValueParser).
func (BoolLiteral) String ¶
func (b BoolLiteral) String() string
func (BoolLiteral) Type ¶
func (b BoolLiteral) Type() Type
func (*BoolLiteral) UnmarshalBinary ¶
func (b *BoolLiteral) UnmarshalBinary(data []byte) error
func (BoolLiteral) Value ¶
func (b BoolLiteral) Value() bool
type BooleanExprVisitor ¶
type BooleanExprVisitor[T any] interface { VisitTrue() T VisitFalse() T VisitNot(childResult T) T VisitAnd(left, right T) T VisitOr(left, right T) T VisitUnbound(UnboundPredicate) T VisitBound(BoundPredicate) T }
BooleanExprVisitor is an interface for recursively visiting the nodes of a boolean expression
type BooleanExpression ¶
type BooleanExpression interface {
fmt.Stringer
Op() Operation
Negate() BooleanExpression
Equals(BooleanExpression) bool
}
BooleanExpression represents a full expression which will evaluate to a boolean value such as GreaterThan or StartsWith, etc.
func BindExpr ¶
func BindExpr(s *Schema, expr BooleanExpression, caseSensitive bool) (BooleanExpression, error)
BindExpr recursively binds each portion of an expression using the provided schema. Because the expression can end up being simplified to just AlwaysTrue/AlwaysFalse, this returns a BooleanExpression.
func IsIn ¶
func IsIn[T LiteralType](t UnboundTerm, vals ...T) BooleanExpression
IsIn is a convenience wrapper for constructing an unbound set predicate for OpIn. It returns a BooleanExpression instead of an UnboundPredicate because depending on the arguments, it can automatically reduce to AlwaysFalse or AlwaysTrue (if given no values for examples). It may also reduce to EqualTo if only one value is provided.
Will panic if t is nil
func NewAnd ¶
func NewAnd(left, right BooleanExpression, addl ...BooleanExpression) BooleanExpression
NewAnd will construct a new AndExpr, allowing the caller to provide potentially more than just two arguments which will be folded to create an appropriate expression tree. i.e. NewAnd(a, b, c, d) becomes AndExpr(a, AndExpr(b, AndExpr(c, d)))
Slight optimizations are performed on creation if either argument is AlwaysFalse or AlwaysTrue by performing reductions. If any argument is AlwaysFalse, then everything will get folded to a return of AlwaysFalse. If an argument is AlwaysTrue, then the other argument will be returned directly rather than creating an AndExpr.
Will panic if any argument is nil
func NewNot ¶
func NewNot(child BooleanExpression) BooleanExpression
NewNot creates a BooleanExpression representing a "Not" operation on the given argument. It will optimize slightly though:
If the argument is AlwaysTrue or AlwaysFalse, the appropriate inverse expression will be returned directly. If the argument is itself a NotExpr, then the child will be returned rather than NotExpr(NotExpr(child)).
func NewOr ¶
func NewOr(left, right BooleanExpression, addl ...BooleanExpression) BooleanExpression
NewOr will construct a new OrExpr, allowing the caller to provide potentially more than just two arguments which will be folded to create an appropriate expression tree. i.e. NewOr(a, b, c, d) becomes OrExpr(a, OrExpr(b, OrExpr(c, d)))
Slight optimizations are performed on creation if either argument is AlwaysFalse or AlwaysTrue by performing reductions. If any argument is AlwaysTrue, then everything will get folded to a return of AlwaysTrue. If an argument is AlwaysFalse, then the other argument will be returned directly rather than creating an OrExpr.
Will panic if any argument is nil
func NotIn ¶
func NotIn[T LiteralType](t UnboundTerm, vals ...T) BooleanExpression
NotIn is a convenience wrapper for constructing an unbound set predicate for OpNotIn. It returns a BooleanExpression instead of an UnboundPredicate because depending on the arguments, it can automatically reduce to AlwaysFalse or AlwaysTrue (if given no values for examples). It may also reduce to NotEqualTo if only one value is provided.
Will panic if t is nil
func ParseExpr ¶ added in v0.7.0
func ParseExpr(data []byte, schema *Schema) (BooleanExpression, error)
ParseExpr parses an expression from its REST JSON form (a request "filter" or a task's residual filter).
With a schema, literals take the referenced field's type (e.g. "2022-08-14" on a date column becomes a DateLiteral) and field names are resolved case-sensitively, since REST field names are authoritative. Without a schema literals fall back to the base JSON kind: Int64Literal, Float64Literal, StringLiteral, or BoolLiteral. The result is unbound.
Transform terms (e.g. {"type":"transform","transform":"bucket[16]","term":"id"}) parse into an UnboundTransform. The term resolves its result type against the schema, and a full predicate over a transform term can be bound and evaluated against rows.
func RewriteNotExpr ¶
func RewriteNotExpr(expr BooleanExpression) (BooleanExpression, error)
RewriteNotExpr rewrites a boolean expression to remove "Not" nodes from the expression tree. This is because Projections assume there are no "not" nodes.
Not nodes will be replaced with simply calling `Negate` on the child in the tree.
func SanitizeExpression ¶ added in v0.7.0
func SanitizeExpression(expr BooleanExpression) (BooleanExpression, error)
SanitizeExpression returns a copy of expr with every predicate literal replaced by an opaque placeholder while preserving the boolean structure, column references, and operations. It lets a filter be emitted somewhere untrusted (e.g. a metrics ScanReport shipped to a REST sink) without leaking the literal values a user scanned with, mirroring the intent of Java's ExpressionUtil.sanitize.
The only guarantee is that no user value leaks: the exact placeholder text is not part of the contract and may change. In particular, unlike Java — which substitutes type-shaped placeholders (e.g. "(2-digit-int)") — every literal is masked with the same marker, since the goal is to withhold the values, not to preserve their shape or stay structurally comparable across clients. Set predicates (IN / NOT IN) keep their arity so the operation is not misrepresented, but the members are masked.
func SetPredicate ¶
func SetPredicate(op Operation, t UnboundTerm, lits []Literal) BooleanExpression
SetPredicate creates a boolean expression representing a predicate that uses a set of literals as the argument, like In or NotIn. Duplicate literals will be folded into a set, only maintaining the unique literals.
Will panic if op is not a valid Set operation
func TranslateColumnNames ¶
func TranslateColumnNames(expr BooleanExpression, fileSchema *Schema) (BooleanExpression, error)
TranslateColumnNames converts the names of columns in an expression by looking up the field IDs in the file schema. If columns don't exist they are replaced with AlwaysFalse or AlwaysTrue depending on the operator.
type BooleanType ¶
type BooleanType struct{}
func (BooleanType) Equals ¶
func (BooleanType) Equals(other Type) bool
func (BooleanType) String ¶
func (BooleanType) String() string
func (BooleanType) Type ¶
func (BooleanType) Type() string
type BoundBBoxPredicate ¶ added in v0.7.0
type BoundBBoxPredicate interface {
BoundPredicate
BBox() BoundingBox
}
BoundBBoxPredicate is a bound geospatial predicate that tests whether a geometry/geography column's bounds intersect a query bounding box.
type BoundBooleanExprVisitor ¶
type BoundBooleanExprVisitor[T any] interface { BooleanExprVisitor[T] VisitIn(BoundTerm, Set[Literal]) T VisitNotIn(BoundTerm, Set[Literal]) T VisitIsNan(BoundTerm) T VisitNotNan(BoundTerm) T VisitIsNull(BoundTerm) T VisitNotNull(BoundTerm) T VisitEqual(BoundTerm, Literal) T VisitNotEqual(BoundTerm, Literal) T VisitGreaterEqual(BoundTerm, Literal) T VisitGreater(BoundTerm, Literal) T VisitLessEqual(BoundTerm, Literal) T VisitLess(BoundTerm, Literal) T VisitStartsWith(BoundTerm, Literal) T VisitNotStartsWith(BoundTerm, Literal) T }
BoundBooleanExprVisitor builds on BooleanExprVisitor by adding interface methods for visiting bound expressions, because we do casting of literals during binding you can assume that the BoundTerm and the Literal passed to a method have the same type. Sets passed to VisitIn and VisitNotIn are detached when dispatched through VisitBoundPredicate. The internal VisitBoundPredicateRef path may pass borrowed sets to trusted built-in visitors, which must not call Add or mutate their literal values.
type BoundExtract ¶ added in v0.7.0
type BoundExtract interface {
BoundTerm
Path() string
// ExtractValue navigates v to this term's path and casts the leaf to the target type.
ExtractValue(v variant.Value) (Literal, bool)
}
BoundExtract is a bound variant sub-path term used for metrics pruning and residual evaluation.
type BoundGeospatialExprVisitor ¶ added in v0.7.0
type BoundGeospatialExprVisitor[T any] interface { VisitBBoxIntersects(BoundTerm, BoundingBox) T VisitBBoxNotIntersects(BoundTerm, BoundingBox) T }
BoundGeospatialExprVisitor is an optional extension a BoundBooleanExprVisitor may implement to handle the geospatial bbox predicates (BBoxIntersects / BBoxNotIntersects). It is deliberately kept out of BoundBooleanExprVisitor: adding methods to that interface would break every existing implementation - including external ones (query-engine adapters, downstream visitors) - which would then fail to compile until they grew geo methods they may not care about. This mirrors Java, which dispatches geospatial predicates outside its BoundExpressionVisitor. VisitBoundPredicate routes a bbox predicate here only when the visitor implements this interface; a visitor that does not is treated as not supporting geospatial predicates.
type BoundLiteralPredicate ¶
type BoundLiteralPredicate interface {
BoundPredicate
Literal() Literal
AsUnbound(Reference, Literal) UnboundPredicate
}
BoundLiteralPredicate represents a bound boolean expression that utilizes a single literal as an argument, such as Equals or StartsWith.
type BoundPredicate ¶
type BoundPredicate interface {
BooleanExpression
Ref() BoundReference
Term() BoundTerm
}
BoundPredicate is a boolean predicate expression which has been bound to a schema. The underlying reference and term can be retrieved from it.
type BoundReference ¶
type BoundReference interface {
BoundTerm
Field() NestedField
Pos() int
PosPath() []int
}
BoundReference is a named reference that has been bound to a particular field in a given schema.
type BoundSetPredicate ¶
type BoundSetPredicate interface {
BoundPredicate
Literals() Set[Literal]
AsUnbound(Reference, []Literal) UnboundPredicate
}
BoundSetPredicate is a bound expression that utilizes a set of literals such as In or NotIn
type BoundTerm ¶
type BoundTerm interface {
Term
Equals(BoundTerm) bool
Ref() BoundReference
Type() Type
// contains filtered or unexported methods
}
BoundTerm is a simple expression (typically a reference) that evaluates to a value and has been bound to a schema.
type BoundTransform ¶
type BoundTransform struct {
// contains filtered or unexported fields
}
func (*BoundTransform) Equals ¶
func (b *BoundTransform) Equals(other BoundTerm) bool
func (*BoundTransform) MarshalJSON ¶ added in v0.7.0
func (b *BoundTransform) MarshalJSON() ([]byte, error)
func (*BoundTransform) Ref ¶
func (b *BoundTransform) Ref() BoundReference
func (*BoundTransform) String ¶
func (b *BoundTransform) String() string
func (*BoundTransform) Type ¶
func (b *BoundTransform) Type() Type
type BoundUnaryPredicate ¶
type BoundUnaryPredicate interface {
BoundPredicate
AsUnbound(Reference) UnboundPredicate
}
BoundUnaryPredicate is a bound predicate expression that has no arguments
type BoundingBox ¶ added in v0.7.0
type BoundingBox struct {
MinX, MinY, MaxX, MaxY float64
}
BoundingBox is a planar (XY) query bounding box used by the geospatial BBoxIntersects predicate. MinX/MinY are the lower-left corner and MaxX/MaxY the upper-right corner (X is longitude/easting, Y is latitude/northing).
The box is two-dimensional: pruning only compares the X and Y extents of a geometry column's bounds, which is all the Iceberg spec requires for bbox-based data skipping. Z/M extents in a column's bounds are ignored.
Z/M are intentionally omitted for now rather than reserved as fields: XY is sufficient for spec-compliant pruning, and adding optional MinZ/MaxZ/MinM/MaxM later is source-additive. The forward-compat contract is that Valid and Equals are defined over whichever extents the box carries - today X and Y - so adding higher dimensions later extends, rather than reinterprets, existing XY boxes.
func (BoundingBox) Equals ¶ added in v0.7.0
func (b BoundingBox) Equals(other BoundingBox) bool
Equals reports whether two bounding boxes have identical extents.
func (BoundingBox) String ¶ added in v0.7.0
func (b BoundingBox) String() string
func (BoundingBox) Valid ¶ added in v0.7.0
func (b BoundingBox) Valid() bool
Valid reports whether the box is well-formed: no coordinate is NaN and the minimum of each axis does not exceed its maximum. An infinite bound is allowed (an open half-plane). An invalid box would silently mis-prune - a NaN makes every intersection test false (pruning matching files), and an inverted box (min > max) reports the wrong overlap - so BBoxIntersects rejects one.
type BucketTransform ¶
type BucketTransform struct {
NumBuckets int
}
BucketTransform transforms values into a bucket partition value. It is parameterized by a number of buckets. Bucket partition transforms use a 32-bit hash of the source value to produce a positive value by mod the bucket number.
func (BucketTransform) Apply ¶
func (t BucketTransform) Apply(value Optional[Literal]) Optional[Literal]
func (BucketTransform) CanTransform ¶ added in v0.4.0
func (BucketTransform) CanTransform(t Type) bool
func (BucketTransform) Equals ¶
func (t BucketTransform) Equals(other Transform) bool
func (BucketTransform) MarshalText ¶
func (t BucketTransform) MarshalText() ([]byte, error)
func (BucketTransform) PreservesOrder ¶ added in v0.2.0
func (BucketTransform) PreservesOrder() bool
func (BucketTransform) Project ¶
func (t BucketTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (BucketTransform) ResultType ¶
func (BucketTransform) ResultType(Type) Type
func (BucketTransform) String ¶
func (t BucketTransform) String() string
func (BucketTransform) ToHumanStr ¶ added in v0.2.0
func (BucketTransform) ToHumanStr(val any) string
func (BucketTransform) ToHumanStrType ¶ added in v0.7.0
func (t BucketTransform) ToHumanStrType(_ Type, val any) string
func (BucketTransform) Transformer ¶
func (t BucketTransform) Transformer(src Type) func(any) Optional[int32]
type Comparator ¶
type Comparator[T LiteralType] func(v1, v2 T) int
Comparator is a comparison function for specific literal types:
returns 0 if v1 == v2 returns <0 if v1 < v2 returns >0 if v1 > v2
type DataFile ¶
type DataFile interface {
// ContentType is the type of the content stored by the data file,
// either Data, Equality deletes, or Position deletes. All v1 files
// are Data files.
ContentType() ManifestEntryContent
// FilePath is the full URI for the file, complete with FS scheme.
FilePath() string
// FileFormat is the format of the data file, AVRO, Orc, or Parquet.
FileFormat() FileFormat
// Partition returns a mapping of field id to partition value for
// each of the partition spec's fields.
Partition() map[int]any
// Count returns the number of records in this file.
Count() int64
// FileSizeBytes is the total file size in bytes.
FileSizeBytes() int64
// ColumnSizes is a mapping from column id to the total size on disk
// of all regions that store the column. Does not include bytes
// necessary to read other columns, like footers. Map will be nil for
// row-oriented formats (avro).
ColumnSizes() map[int]int64
// ValueCounts is a mapping from column id to the number of values
// in the column, including null and NaN values.
ValueCounts() map[int]int64
// NullValueCounts is a mapping from column id to the number of
// null values in the column.
NullValueCounts() map[int]int64
// NaNValueCounts is a mapping from column id to the number of NaN
// values in the column.
NaNValueCounts() map[int]int64
// DistictValueCounts is a mapping from column id to the number of
// distinct values in the column. Distinct counts must be derived
// using values in the file by counting or using sketches, but not
// using methods like merging existing distinct counts.
DistinctValueCounts() map[int]int64
// LowerBoundValues is a mapping from column id to the lower bounded
// value of the column, serialized as binary. Each value in the column
// must be less than or requal to all non-null, non-NaN values in the
// column for the file.
LowerBoundValues() map[int][]byte
// UpperBoundValues is a mapping from column id to the upper bounded
// value of the column, serialized as binary. Each value in the column
// must be greater than or equal to all non-null, non-NaN values in
// the column for the file.
UpperBoundValues() map[int][]byte
// KeyMetadata is implementation-specific key metadata for encryption.
KeyMetadata() []byte
// SplitOffsets are the split offsets for the data file. For example,
// all row group offsets in a Parquet file. Must be sorted ascending.
SplitOffsets() []int64
// EqualityFieldIDs are used to determine row equality in equality
// delete files. It is required when the content type is
// EntryContentEqDeletes.
EqualityFieldIDs() []int
// SortOrderID returns the id representing the sort order for this
// file, or nil if there is no sort order.
SortOrderID() *int
// SpecID returns the partition spec id for this data file, inherited
// from the manifest that the data file was read from
SpecID() int32
// FirstRowID returns the first row ID for this data file ( v3+ only )
FirstRowID() *int64
// ReferencedDataFile returns the location of the data file that deletion vector reference
ReferencedDataFile() *string
// ContentOffset returns the offset in the file where the content starts ( v3+ only )
ContentOffset() *int64
// ContentSizeInBytes returns the length of referenced contented stored in the file (v3+ only)
ContentSizeInBytes() *int64
}
DataFile is the interface for reading the information about a given data file indicated by an entry in a manifest list.
func DataFileWithoutColumnStats ¶ added in v0.7.0
DataFileWithoutColumnStats returns a copy of the built-in DataFile with transient column statistics removed. Other DataFile implementations are returned unchanged because the package cannot safely clone their private state. The returned value is for read-only planning and must not be passed to a ManifestWriter, because the omitted statistics cannot be recovered. ManifestWriter and the DataFile Avro codec reject the copy with ErrInvalidArgument.
type DataFileBuilder ¶
type DataFileBuilder struct {
// contains filtered or unexported fields
}
DataFileBuilder is a helper for building a data file struct which will conform to the DataFile interface.
func NewDataFileBuilder ¶
func NewDataFileBuilder( spec PartitionSpec, content ManifestEntryContent, path string, format FileFormat, fieldIDToPartitionData map[int]any, fieldIDToLogicalType map[int]string, fieldIDToFixedSize map[int]int, recordCount int64, fileSize int64, ) (*DataFileBuilder, error)
NewDataFileBuilder is passed all of the required fields and then allows all of the optional fields to be set by calling the corresponding methods before calling DataFileBuilder.Build to construct the object. The fieldIDToFixedSize argument is retained for source compatibility and is ignored; manifest schemas determine decimal fixed widths during encoding. Byte-slice partition values are cloned. Other partition values must be the immutable scalar or literal values produced by Iceberg's manifest decoder; mutable caller-defined values are retained by reference.
func (*DataFileBuilder) BlockSizeInBytes ¶
func (b *DataFileBuilder) BlockSizeInBytes(size int64) *DataFileBuilder
BlockSizeInBytes sets the block size in bytes for the data file. Deprecated in v2.
func (*DataFileBuilder) Build ¶
func (b *DataFileBuilder) Build() DataFile
func (*DataFileBuilder) ColumnSizes ¶
func (b *DataFileBuilder) ColumnSizes(sizes map[int]int64) *DataFileBuilder
ColumnSizes sets the column sizes for the data file.
func (*DataFileBuilder) ContentOffset ¶ added in v0.4.0
func (b *DataFileBuilder) ContentOffset(offset int64) *DataFileBuilder
func (*DataFileBuilder) ContentSizeInBytes ¶ added in v0.4.0
func (b *DataFileBuilder) ContentSizeInBytes(size int64) *DataFileBuilder
func (*DataFileBuilder) DistinctValueCounts
deprecated
func (b *DataFileBuilder) DistinctValueCounts(counts map[int]int64) *DataFileBuilder
DistinctValueCounts sets the distinct value counts for the data file.
Deprecated: distinct_counts (field 111) is deprecated in every version of the Iceberg spec (apache/iceberg#12182). The Avro manifest-entry schemas omit the field for v1, v2, and v3, so values set here are not transported in manifests written by this library. The setter is retained for round-tripping legacy DataFiles read from older manifests; new code should not call it.
func (*DataFileBuilder) EqualityFieldIDs ¶
func (b *DataFileBuilder) EqualityFieldIDs(ids []int) *DataFileBuilder
EqualityFieldIDs sets the equality field ids for the data file.
func (*DataFileBuilder) FirstRowID ¶ added in v0.4.0
func (b *DataFileBuilder) FirstRowID(id int64) *DataFileBuilder
func (*DataFileBuilder) KeyMetadata ¶
func (b *DataFileBuilder) KeyMetadata(key []byte) *DataFileBuilder
KeyMetadata sets the key metadata for the data file.
func (*DataFileBuilder) LowerBoundValues ¶
func (b *DataFileBuilder) LowerBoundValues(bounds map[int][]byte) *DataFileBuilder
LowerBoundValues sets the lower bound values for the data file.
func (*DataFileBuilder) NaNValueCounts ¶
func (b *DataFileBuilder) NaNValueCounts(counts map[int]int64) *DataFileBuilder
NaNValueCounts sets the NaN value counts for the data file.
func (*DataFileBuilder) NullValueCounts ¶
func (b *DataFileBuilder) NullValueCounts(counts map[int]int64) *DataFileBuilder
NullValueCounts sets the null value counts for the data file.
func (*DataFileBuilder) ReferencedDataFile ¶ added in v0.4.0
func (b *DataFileBuilder) ReferencedDataFile(path string) *DataFileBuilder
func (*DataFileBuilder) SortOrderID ¶
func (b *DataFileBuilder) SortOrderID(id int) *DataFileBuilder
SortOrderID sets the sort order id for the data file.
func (*DataFileBuilder) SplitOffsets ¶
func (b *DataFileBuilder) SplitOffsets(offsets []int64) *DataFileBuilder
SplitOffsets sets the split offsets for the data file.
func (*DataFileBuilder) UpperBoundValues ¶
func (b *DataFileBuilder) UpperBoundValues(bounds map[int][]byte) *DataFileBuilder
UpperBoundValues sets the upper bound values for the data file.
func (*DataFileBuilder) ValueCounts ¶
func (b *DataFileBuilder) ValueCounts(counts map[int]int64) *DataFileBuilder
ValueCounts sets the value counts for the data file.
type DateLiteral ¶
type DateLiteral Date
func (DateLiteral) Any ¶ added in v0.2.0
func (d DateLiteral) Any() any
func (DateLiteral) Comparator ¶
func (DateLiteral) Comparator() Comparator[Date]
func (DateLiteral) Decrement ¶
func (d DateLiteral) Decrement() Literal
func (DateLiteral) Equals ¶
func (d DateLiteral) Equals(other Literal) bool
func (DateLiteral) Increment ¶
func (d DateLiteral) Increment() Literal
func (DateLiteral) MarshalBinary ¶
func (d DateLiteral) MarshalBinary() (data []byte, err error)
func (DateLiteral) MarshalJSON ¶ added in v0.7.0
func (l DateLiteral) MarshalJSON() ([]byte, error)
func (DateLiteral) String ¶
func (d DateLiteral) String() string
func (DateLiteral) Type ¶
func (d DateLiteral) Type() Type
func (*DateLiteral) UnmarshalBinary ¶
func (d *DateLiteral) UnmarshalBinary(data []byte) error
func (DateLiteral) Value ¶
func (d DateLiteral) Value() Date
type DateType ¶
type DateType struct{}
DateType represents a calendar date without a timezone or time, represented as a 32-bit integer denoting the number of days since the unix epoch.
type DayTransform ¶
type DayTransform struct{}
DayTransform transforms a datetime value into a date value.
func (DayTransform) Apply ¶
func (DayTransform) Apply(value Optional[Literal]) (out Optional[Literal])
func (DayTransform) CanTransform ¶ added in v0.4.0
func (t DayTransform) CanTransform(sourceType Type) bool
func (DayTransform) Equals ¶
func (DayTransform) Equals(other Transform) bool
func (DayTransform) MarshalText ¶
func (t DayTransform) MarshalText() ([]byte, error)
func (DayTransform) PreservesOrder ¶ added in v0.2.0
func (DayTransform) PreservesOrder() bool
func (DayTransform) Project ¶
func (t DayTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (DayTransform) ResultType ¶
func (DayTransform) ResultType(Type) Type
func (DayTransform) String ¶
func (DayTransform) String() string
func (DayTransform) ToHumanStr ¶ added in v0.2.0
func (DayTransform) ToHumanStr(val any) string
func (DayTransform) ToHumanStrType ¶ added in v0.7.0
func (t DayTransform) ToHumanStrType(_ Type, val any) string
func (DayTransform) Transformer ¶
type Decimal ¶
type Decimal struct {
Val decimal.Decimal128
Scale int
}
type DecimalLiteral ¶
type DecimalLiteral Decimal
func (DecimalLiteral) Any ¶ added in v0.2.0
func (d DecimalLiteral) Any() any
func (DecimalLiteral) Comparator ¶
func (DecimalLiteral) Comparator() Comparator[Decimal]
func (DecimalLiteral) Decrement ¶
func (d DecimalLiteral) Decrement() Literal
func (DecimalLiteral) Equals ¶
func (d DecimalLiteral) Equals(other Literal) bool
func (DecimalLiteral) Increment ¶
func (d DecimalLiteral) Increment() Literal
func (DecimalLiteral) MarshalBinary ¶
func (d DecimalLiteral) MarshalBinary() (data []byte, err error)
func (DecimalLiteral) MarshalJSON ¶ added in v0.7.0
func (l DecimalLiteral) MarshalJSON() ([]byte, error)
func (DecimalLiteral) String ¶
func (d DecimalLiteral) String() string
func (DecimalLiteral) Type ¶
func (d DecimalLiteral) Type() Type
Type returns a DecimalType built from the literal's scale and a hardcoded precision of 9. The precision is NOT the originating column's declared precision; DecimalLiteral does not carry precision. Callers that need the real column precision must consult the bound field's type rather than lit.Type(). See https://github.com/apache/iceberg-go/issues/1028.
func (*DecimalLiteral) UnmarshalBinary ¶
func (d *DecimalLiteral) UnmarshalBinary(data []byte) error
func (DecimalLiteral) Value ¶
func (d DecimalLiteral) Value() Decimal
type DecimalType ¶
type DecimalType struct {
// contains filtered or unexported fields
}
func DecimalTypeOf ¶
func DecimalTypeOf(prec, scale int) DecimalType
func (DecimalType) Equals ¶
func (d DecimalType) Equals(other Type) bool
func (DecimalType) Precision ¶
func (d DecimalType) Precision() int
func (DecimalType) Scale ¶
func (d DecimalType) Scale() int
func (DecimalType) String ¶
func (d DecimalType) String() string
func (DecimalType) Type ¶
func (d DecimalType) Type() string
type FieldLabel ¶ added in v0.7.0
type FieldLabel struct {
FieldID int `json:"field-id"`
Labels Properties `json:"labels,omitempty"`
}
FieldLabel holds the labels for a single schema field. FieldID may reference a column that has since been dropped, so callers must tolerate unresolved IDs.
type FieldSummary ¶
type FileFormat ¶
type FileFormat string
FileFormat defines constants for the format of data files.
const ( AvroFile FileFormat = "AVRO" OrcFile FileFormat = "ORC" ParquetFile FileFormat = "PARQUET" PuffinFile FileFormat = "PUFFIN" )
func FileFormatFromString ¶ added in v0.6.0
func FileFormatFromString(s string) (FileFormat, error)
FileFormatFromString parses a file format string (case-insensitive).
type FixedLiteral ¶
type FixedLiteral []byte
func (FixedLiteral) Any ¶ added in v0.2.0
func (f FixedLiteral) Any() any
func (FixedLiteral) Comparator ¶
func (FixedLiteral) Comparator() Comparator[[]byte]
func (FixedLiteral) Equals ¶
func (f FixedLiteral) Equals(other Literal) bool
func (FixedLiteral) MarshalBinary ¶
func (f FixedLiteral) MarshalBinary() (data []byte, err error)
func (FixedLiteral) MarshalJSON ¶ added in v0.7.0
func (l FixedLiteral) MarshalJSON() ([]byte, error)
func (FixedLiteral) String ¶
func (f FixedLiteral) String() string
func (FixedLiteral) Type ¶
func (f FixedLiteral) Type() Type
func (*FixedLiteral) UnmarshalBinary ¶
func (f *FixedLiteral) UnmarshalBinary(data []byte) error
func (FixedLiteral) Value ¶
func (f FixedLiteral) Value() []byte
type FixedType ¶
type FixedType struct {
// contains filtered or unexported fields
}
func FixedTypeOf ¶
type Float32Literal ¶
type Float32Literal float32
func (Float32Literal) Any ¶ added in v0.2.0
func (f Float32Literal) Any() any
func (Float32Literal) Comparator ¶
func (Float32Literal) Comparator() Comparator[float32]
func (Float32Literal) Equals ¶
func (f Float32Literal) Equals(other Literal) bool
func (Float32Literal) MarshalBinary ¶
func (f Float32Literal) MarshalBinary() (data []byte, err error)
func (Float32Literal) MarshalJSON ¶ added in v0.7.0
func (l Float32Literal) MarshalJSON() ([]byte, error)
func (Float32Literal) String ¶
func (f Float32Literal) String() string
func (Float32Literal) Type ¶
func (f Float32Literal) Type() Type
func (*Float32Literal) UnmarshalBinary ¶
func (f *Float32Literal) UnmarshalBinary(data []byte) error
func (Float32Literal) Value ¶
func (f Float32Literal) Value() float32
type Float32Type ¶
type Float32Type struct{}
Float32Type is the "float" type in the iceberg spec.
func (Float32Type) Equals ¶
func (Float32Type) Equals(other Type) bool
func (Float32Type) String ¶
func (Float32Type) String() string
func (Float32Type) Type ¶
func (Float32Type) Type() string
type Float64Literal ¶
type Float64Literal float64
func (Float64Literal) Any ¶ added in v0.2.0
func (f Float64Literal) Any() any
func (Float64Literal) Comparator ¶
func (Float64Literal) Comparator() Comparator[float64]
func (Float64Literal) Equals ¶
func (f Float64Literal) Equals(other Literal) bool
func (Float64Literal) MarshalBinary ¶
func (f Float64Literal) MarshalBinary() (data []byte, err error)
func (Float64Literal) MarshalJSON ¶ added in v0.7.0
func (l Float64Literal) MarshalJSON() ([]byte, error)
func (Float64Literal) String ¶
func (f Float64Literal) String() string
func (Float64Literal) Type ¶
func (f Float64Literal) Type() Type
func (*Float64Literal) UnmarshalBinary ¶
func (f *Float64Literal) UnmarshalBinary(data []byte) error
func (Float64Literal) Value ¶
func (f Float64Literal) Value() float64
type Float64Type ¶
type Float64Type struct{}
Float64Type represents the "double" type of the iceberg spec.
func (Float64Type) Equals ¶
func (Float64Type) Equals(other Type) bool
func (Float64Type) String ¶
func (Float64Type) String() string
func (Float64Type) Type ¶
func (Float64Type) Type() string
type GeoLiteral ¶ added in v0.7.0
type GeoLiteral struct {
// contains filtered or unexported fields
}
GeoLiteral is a non-null geometry or geography single-value bound. Iceberg serializes geospatial bounds as the concatenation of little-endian float64 coordinates in X, Y[, Z][, M] order (spec Appendix D), not as WKB. Those coordinates have no total order, so - unlike every other literal - GeoLiteral is deliberately not a TypedLiteral: it exposes no Comparator and refuses to be cast to an orderable type. That keeps ordering-based pruning from silently running bytes.Compare over coordinate bytes (which would yield wrong answers) and makes it error instead. Type() reports the concrete GeometryType or GeographyType, so callers that key off the literal's own type - pruning, To, logging - see the truth rather than plain binary.
func (GeoLiteral) Any ¶ added in v0.7.0
func (g GeoLiteral) Any() any
func (GeoLiteral) Equals ¶ added in v0.7.0
func (g GeoLiteral) Equals(other Literal) bool
func (GeoLiteral) MarshalBinary ¶ added in v0.7.0
func (g GeoLiteral) MarshalBinary() ([]byte, error)
func (GeoLiteral) String ¶ added in v0.7.0
func (g GeoLiteral) String() string
func (GeoLiteral) Type ¶ added in v0.7.0
func (g GeoLiteral) Type() Type
func (GeoLiteral) Value ¶ added in v0.7.0
func (g GeoLiteral) Value() []byte
type GeographyType ¶ added in v0.7.0
type GeographyType struct {
// contains filtered or unexported fields
}
func GeographyTypeOf ¶ added in v0.7.0
func GeographyTypeOf(crs string, algorithm string) (GeographyType, error)
func (GeographyType) Algorithm ¶ added in v0.7.0
func (g GeographyType) Algorithm() string
func (GeographyType) CRS ¶ added in v0.7.0
func (g GeographyType) CRS() string
func (GeographyType) Equals ¶ added in v0.7.0
func (g GeographyType) Equals(other Type) bool
func (GeographyType) String ¶ added in v0.7.0
func (g GeographyType) String() string
func (GeographyType) Type ¶ added in v0.7.0
func (g GeographyType) Type() string
type GeometryType ¶ added in v0.7.0
type GeometryType struct {
// contains filtered or unexported fields
}
func GeometryTypeOf ¶ added in v0.7.0
func GeometryTypeOf(crs string) (GeometryType, error)
func (GeometryType) CRS ¶ added in v0.7.0
func (g GeometryType) CRS() string
func (GeometryType) Equals ¶ added in v0.7.0
func (g GeometryType) Equals(other Type) bool
func (GeometryType) String ¶ added in v0.7.0
func (g GeometryType) String() string
func (GeometryType) Type ¶ added in v0.7.0
func (g GeometryType) Type() string
type HourTransform ¶
type HourTransform struct{}
HourTransform transforms a datetime value into an hour value.
func (HourTransform) Apply ¶
func (HourTransform) Apply(value Optional[Literal]) (out Optional[Literal])
func (HourTransform) CanTransform ¶ added in v0.4.0
func (t HourTransform) CanTransform(sourceType Type) bool
func (HourTransform) Equals ¶
func (HourTransform) Equals(other Transform) bool
func (HourTransform) MarshalText ¶
func (t HourTransform) MarshalText() ([]byte, error)
func (HourTransform) PreservesOrder ¶ added in v0.2.0
func (HourTransform) PreservesOrder() bool
func (HourTransform) Project ¶
func (t HourTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (HourTransform) ResultType ¶
func (HourTransform) ResultType(Type) Type
func (HourTransform) String ¶
func (HourTransform) String() string
func (HourTransform) ToHumanStr ¶ added in v0.2.0
func (HourTransform) ToHumanStr(val any) string
func (HourTransform) ToHumanStrType ¶ added in v0.7.0
func (t HourTransform) ToHumanStrType(_ Type, val any) string
func (HourTransform) Transformer ¶
type IdentityTransform ¶
type IdentityTransform struct{}
IdentityTransform uses the identity function, performing no transformation but instead partitioning on the value itself.
func (IdentityTransform) Apply ¶
func (IdentityTransform) Apply(value Optional[Literal]) Optional[Literal]
func (IdentityTransform) CanTransform ¶ added in v0.4.0
func (IdentityTransform) CanTransform(t Type) bool
func (IdentityTransform) Equals ¶
func (IdentityTransform) Equals(other Transform) bool
func (IdentityTransform) MarshalText ¶
func (t IdentityTransform) MarshalText() ([]byte, error)
func (IdentityTransform) PreservesOrder ¶ added in v0.2.0
func (IdentityTransform) PreservesOrder() bool
func (IdentityTransform) Project ¶
func (t IdentityTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (IdentityTransform) ResultType ¶
func (IdentityTransform) ResultType(t Type) Type
func (IdentityTransform) String ¶
func (IdentityTransform) String() string
func (IdentityTransform) ToHumanStr ¶ added in v0.2.0
func (IdentityTransform) ToHumanStr(val any) string
func (IdentityTransform) ToHumanStrType ¶ added in v0.7.0
func (t IdentityTransform) ToHumanStrType(typ Type, val any) string
ToHumanStrType is the type-aware form invoked by PartitionToPath. It appends "+00:00" for TimestampTz/TimestampTzNs — information the value-only ToHumanStr(any) cannot recover by design
type Int32Literal ¶
type Int32Literal int32
func (Int32Literal) Any ¶ added in v0.2.0
func (i Int32Literal) Any() any
func (Int32Literal) Comparator ¶
func (Int32Literal) Comparator() Comparator[int32]
func (Int32Literal) Decrement ¶
func (i Int32Literal) Decrement() Literal
func (Int32Literal) Equals ¶
func (i Int32Literal) Equals(other Literal) bool
func (Int32Literal) Increment ¶
func (i Int32Literal) Increment() Literal
func (Int32Literal) MarshalBinary ¶
func (i Int32Literal) MarshalBinary() (data []byte, err error)
func (Int32Literal) MarshalJSON ¶ added in v0.7.0
func (l Int32Literal) MarshalJSON() ([]byte, error)
func (Int32Literal) String ¶
func (i Int32Literal) String() string
func (Int32Literal) Type ¶
func (i Int32Literal) Type() Type
func (*Int32Literal) UnmarshalBinary ¶
func (i *Int32Literal) UnmarshalBinary(data []byte) error
func (Int32Literal) Value ¶
func (i Int32Literal) Value() int32
type Int64Literal ¶
type Int64Literal int64
func (Int64Literal) Any ¶ added in v0.2.0
func (i Int64Literal) Any() any
func (Int64Literal) Comparator ¶
func (Int64Literal) Comparator() Comparator[int64]
func (Int64Literal) Decrement ¶
func (i Int64Literal) Decrement() Literal
func (Int64Literal) Equals ¶
func (i Int64Literal) Equals(other Literal) bool
func (Int64Literal) Increment ¶
func (i Int64Literal) Increment() Literal
func (Int64Literal) MarshalBinary ¶
func (i Int64Literal) MarshalBinary() (data []byte, err error)
func (Int64Literal) MarshalJSON ¶ added in v0.7.0
func (l Int64Literal) MarshalJSON() ([]byte, error)
func (Int64Literal) String ¶
func (i Int64Literal) String() string
func (Int64Literal) Type ¶
func (i Int64Literal) Type() Type
func (*Int64Literal) UnmarshalBinary ¶
func (i *Int64Literal) UnmarshalBinary(data []byte) error
func (Int64Literal) Value ¶
func (i Int64Literal) Value() int64
type Labels ¶ added in v0.7.0
type Labels struct {
ObjectLabels Properties `json:"object-labels,omitempty"`
Fields []FieldLabel `json:"fields,omitempty"`
}
Labels is catalog-provided enrichment returned with a table or view in a REST load response. Labels are transient: generated per request and never persisted to metadata, so a client may ignore them and two catalogs may return different labels for the same object. It maps to the "labels" property of LoadTableResult and LoadViewResult in the REST catalog spec.
func (*Labels) Field ¶ added in v0.7.0
func (l *Labels) Field(fieldID int) Properties
Field returns the labels for the given field ID, or nil if it carries none (including a dropped field). The first match wins if IDs repeat.
func (*Labels) IsEmpty ¶ added in v0.7.0
IsEmpty reports whether there are neither object-level nor field-level labels. It is nil-safe.
func (*Labels) Object ¶ added in v0.7.0
func (l *Labels) Object() Properties
Object returns the object-level labels, or nil if there are none.
type ListType ¶
type ListType struct {
ElementID int `json:"element-id"`
Element Type `json:"-"`
ElementRequired bool `json:"element-required"`
}
func (*ListType) ElementField ¶
func (l *ListType) ElementField() NestedField
func (*ListType) Fields ¶
func (l *ListType) Fields() []NestedField
func (*ListType) MarshalJSON ¶
func (*ListType) UnmarshalJSON ¶
type Literal ¶
type Literal interface {
fmt.Stringer
encoding.BinaryMarshaler
Any() any
Type() Type
To(Type) (Literal, error)
Equals(Literal) bool
}
Literal is a non-null literal value. It can be casted using To and be checked for equality against other literals.
func CastVariantLiteral ¶ added in v0.7.0
func CastVariantLiteral(v variant.Value, typ PrimitiveType) (Literal, bool)
CastVariantLiteral casts a leaf variant value to typ and wraps it as a Literal. Exported for table/internal; not part of the stable public API.
func Float32AboveMaxLiteral ¶
func Float32AboveMaxLiteral() Literal
func Float32BelowMinLiteral ¶
func Float32BelowMinLiteral() Literal
func Float64AboveMaxLiteral ¶
func Float64AboveMaxLiteral() Literal
func Float64BelowMinLiteral ¶
func Float64BelowMinLiteral() Literal
func Int32AboveMaxLiteral ¶
func Int32AboveMaxLiteral() Literal
func Int32BelowMinLiteral ¶
func Int32BelowMinLiteral() Literal
func Int64AboveMaxLiteral ¶
func Int64AboveMaxLiteral() Literal
func Int64BelowMinLiteral ¶
func Int64BelowMinLiteral() Literal
func LiteralFromBytes ¶
LiteralFromBytes uses the defined Iceberg spec for how to serialize a value of a the provided type and returns the appropriate Literal value from it.
If you already have a value of the desired Literal type, you could alternatively call UnmarshalBinary on it yourself manually.
This is primarily used for retrieving stat values.
func NewLiteral ¶
func NewLiteral[T LiteralType](val T) Literal
NewLiteral provides a literal based on the type of T.
The type switch is performed on &val (a pointer) rather than val itself. Boxing a small scalar such as bool into an interface makes the compiler reference runtime.staticuint64s, which emits a relocation the linker rejects as misaligned on big-endian platforms such as s390x. Switching on a pointer boxes only the pointer, never the scalar, avoiding that reference.
type LiteralType ¶
type LiteralType interface {
bool | int32 | int64 | float32 | float64 | Date |
Time | Timestamp | TimestampNano | string | []byte | uuid.UUID | Decimal |
variant.Value
}
LiteralType is a generic type constraint for the explicit Go types that we allow for literal values. This represents the actual primitive types that exist in Iceberg
type ManifestBuilder ¶ added in v0.2.0
type ManifestBuilder struct {
// contains filtered or unexported fields
}
func NewManifestFile ¶ added in v0.2.0
func (*ManifestBuilder) AddedFiles ¶ added in v0.2.0
func (b *ManifestBuilder) AddedFiles(cnt int32) *ManifestBuilder
func (*ManifestBuilder) AddedRows ¶ added in v0.2.0
func (b *ManifestBuilder) AddedRows(cnt int64) *ManifestBuilder
func (*ManifestBuilder) Build ¶ added in v0.2.0
func (b *ManifestBuilder) Build() ManifestFile
func (*ManifestBuilder) Content ¶ added in v0.2.0
func (b *ManifestBuilder) Content(content ManifestContent) *ManifestBuilder
func (*ManifestBuilder) DeletedFiles ¶ added in v0.2.0
func (b *ManifestBuilder) DeletedFiles(cnt int32) *ManifestBuilder
func (*ManifestBuilder) DeletedRows ¶ added in v0.2.0
func (b *ManifestBuilder) DeletedRows(cnt int64) *ManifestBuilder
func (*ManifestBuilder) ExistingFiles ¶ added in v0.2.0
func (b *ManifestBuilder) ExistingFiles(cnt int32) *ManifestBuilder
func (*ManifestBuilder) ExistingRows ¶ added in v0.2.0
func (b *ManifestBuilder) ExistingRows(cnt int64) *ManifestBuilder
func (*ManifestBuilder) KeyMetadata ¶ added in v0.2.0
func (b *ManifestBuilder) KeyMetadata(km []byte) *ManifestBuilder
func (*ManifestBuilder) Partitions ¶ added in v0.2.0
func (b *ManifestBuilder) Partitions(p []FieldSummary) *ManifestBuilder
func (*ManifestBuilder) SequenceNum ¶ added in v0.2.0
func (b *ManifestBuilder) SequenceNum(num, minSeqNum int64) *ManifestBuilder
type ManifestContent ¶
type ManifestContent int32
ManifestContent indicates the type of data inside of the files described by a manifest. This will indicate whether the data files contain active data or deleted rows.
const ( ManifestContentData ManifestContent = 0 ManifestContentDeletes ManifestContent = 1 )
func (ManifestContent) String ¶ added in v0.2.0
func (m ManifestContent) String() string
type ManifestEntry ¶
type ManifestEntry interface {
// Status returns the type of the file tracked by this entry.
// Whether entries with EntryStatusDELETED are returned depends on the
// caller's discardDeleted option.
Status() ManifestEntryStatus
// SnapshotID is the id where the file was added, or deleted,
// if null it is inherited from the manifest list.
SnapshotID() int64
// SequenceNum returns the data sequence number of the file.
// If it was null and the status is EntryStatusADDED then it
// is inherited from the manifest list.
SequenceNum() int64
// FileSequenceNum returns the file sequence number indicating
// when the file was added. If it was null and the status is
// EntryStatusADDED then it is inherited from the manifest list.
FileSequenceNum() *int64
// DataFile provides the information about the data file indicated
// by this manifest entry.
DataFile() DataFile
// contains filtered or unexported methods
}
ManifestEntry is an interface for both v1 and v2 manifest entries.
func ManifestEntryWithoutColumnStats ¶ added in v0.7.0
func ManifestEntryWithoutColumnStats(entry ManifestEntry) ManifestEntry
ManifestEntryWithoutColumnStats returns a copy of an entry whose built-in DataFile has had transient column statistics removed. The returned entry must not be passed to a ManifestWriter, because omitted statistics would be lost on rewrite. ManifestWriter and the DataFile Avro codec reject entries containing these copies with ErrInvalidArgument.
func NewManifestEntry ¶ added in v0.2.0
func NewManifestEntry(status ManifestEntryStatus, snapshotID *int64, seqNum, fileSeqNum *int64, df DataFile) ManifestEntry
func ReadManifest ¶ added in v0.3.0
func ReadManifest(m ManifestFile, f io.Reader, discardDeleted bool) ([]ManifestEntry, error)
ReadManifest reads in an avro list file and returns a slice of manifest entries or an error if one is encountered. If discardDeleted is true, the returned slice omits entries whose status is "deleted".
type ManifestEntryBuilder ¶ added in v0.2.0
type ManifestEntryBuilder struct {
// contains filtered or unexported fields
}
func NewManifestEntryBuilder ¶ added in v0.2.0
func NewManifestEntryBuilder(status ManifestEntryStatus, snapshotID *int64, data DataFile) *ManifestEntryBuilder
func (*ManifestEntryBuilder) Build ¶ added in v0.2.0
func (b *ManifestEntryBuilder) Build() ManifestEntry
func (*ManifestEntryBuilder) FileSequenceNum ¶ added in v0.2.0
func (b *ManifestEntryBuilder) FileSequenceNum(num int64) *ManifestEntryBuilder
func (*ManifestEntryBuilder) SequenceNum ¶ added in v0.2.0
func (b *ManifestEntryBuilder) SequenceNum(num int64) *ManifestEntryBuilder
type ManifestEntryContent ¶
type ManifestEntryContent int8
ManifestEntryContent defines constants for the type of file contents in the file entries. Data, Position based deletes and equality based deletes.
const ( EntryContentData ManifestEntryContent = 0 EntryContentPosDeletes ManifestEntryContent = 1 EntryContentEqDeletes ManifestEntryContent = 2 )
func (ManifestEntryContent) String ¶
func (m ManifestEntryContent) String() string
type ManifestEntryProjection ¶ added in v0.7.0
type ManifestEntryProjection struct {
IncludePruningStats bool
}
ManifestEntryProjection selects the optional data-file fields decoded while reading a manifest. The fields needed to build a scan task are always read. When IncludePruningStats is true, the metric maps used for pruning are also read: value_counts, null_value_counts, nan_value_counts, lower_bounds, and upper_bounds. column_sizes and deprecated distinct_counts remain omitted.
A projected read is intended for planning paths that use statistics transiently. Callers that need the complete DataFile metadata should use ManifestFile.Entries or ReadManifest instead. Projected entries are read-only planning values and must not be passed to a ManifestWriter, because omitted statistics cannot be recovered on rewrite. ManifestWriter and the DataFile Avro codec reject projected files with ErrInvalidArgument, even when IncludePruningStats is true.
type ManifestEntryStatus ¶
type ManifestEntryStatus int8
ManifestEntryStatus defines constants for the entry status of existing, added or deleted.
const ( EntryStatusEXISTING ManifestEntryStatus = 0 EntryStatusADDED ManifestEntryStatus = 1 EntryStatusDELETED ManifestEntryStatus = 2 )
type ManifestFile ¶
type ManifestFile interface {
// Version returns the version number of this manifest file.
// It must be 1, 2, or 3.
Version() int
// FilePath is the location URI of this manifest file.
FilePath() string
// Length is the length in bytes of the manifest file.
Length() int64
// PartitionSpecID is the ID of the partition spec used to write
// this manifest. It must be listed in the table metadata
// partition-specs.
PartitionSpecID() int32
// ManifestContent is the type of files tracked by this manifest,
// either data or delete files. All v1 manifests track data files.
ManifestContent() ManifestContent
// SnapshotID is the ID of the snapshot where this manifest file
// was added.
SnapshotID() int64
// AddedDataFiles returns the number of entries in the manifest that
// have the status of EntryStatusADDED.
AddedDataFiles() int32
// ExistingDataFiles returns the number of entries in the manifest
// which have the status of EntryStatusEXISTING.
ExistingDataFiles() int32
// DeletedDataFiles returns the number of entries in the manifest
// which have the status of EntryStatusDELETED.
DeletedDataFiles() int32
// AddedRows returns the number of rows in all files of the manifest
// that have status EntryStatusADDED.
AddedRows() int64
// ExistingRows returns the number of rows in all files of the manifest
// which have status EntryStatusEXISTING.
ExistingRows() int64
// DeletedRows returns the number of rows in all files of the manifest
// which have status EntryStatusDELETED.
DeletedRows() int64
// SequenceNum returns the sequence number when this manifest was
// added to the table. Will be 0 for v1 manifest lists.
SequenceNum() int64
// MinSequenceNum is the minimum data sequence number of all live data
// or delete files in the manifest. Will be 0 for v1 manifest lists.
MinSequenceNum() int64
// KeyMetadata returns implementation-specific key metadata for encryption
// if it exists in the manifest list.
KeyMetadata() []byte
// Partitions returns a list of field summaries for each partition
// field in the spec. Each field in the list corresponds to a field in
// the manifest file's partition spec.
Partitions() []FieldSummary
// FirstRowID returns the first _row_id assigned to rows in this manifest (v3+ data manifests only).
// Returns nil for v1/v2 or for delete manifests.
FirstRowID() *int64
// HasAddedFiles returns true if AddedDataFiles > 0 or if it was null.
HasAddedFiles() bool
// HasExistingFiles returns true if ExistingDataFiles > 0 or if it was null.
HasExistingFiles() bool
// Entries streams the manifest entries from the manifest file using
// the provided file system IO interface. Entries that have been
// marked as deleted are skipped if discardDeleted is true.
//
// Prefer Entries over FetchEntries when walking large manifests
// since it avoids loading every entry into memory at once.
//
// Iteration contract:
//
// - On the first error encountered while opening the manifest, decoding
// a record, or applying inheritance, the iterator yields (nil, err)
// and then stops. Callers must treat any non-nil error as terminal and
// break or return without consuming further values.
// - When iteration ends without an error from the read path, the
// iterator may yield a final (nil, closeErr) pair if closing the
// underlying file or manifest reader returns an error. This terminal
// close error is reported only when the consumer ranged through every
// value; an early break suppresses it (see below).
// - Breaking out of the range loop (or any other early termination of
// the yield function) is safe: the iterator releases the underlying
// file handle and reader before returning, and no close error from
// that path is yielded — the caller has already signalled it is no
// longer interested in further values, so an extra synthetic
// (nil, closeErr) tail would be discarded anyway.
Entries(fs iceio.IO, discardDeleted bool) iter.Seq2[ManifestEntry, error]
// FetchEntries reads the manifest list file to fetch the list of
// manifest entries using the provided file system IO interface.
// If discardDeleted is true, entries for files containing deleted rows
// will be skipped.
//
// Deprecated: Use Entries instead, which streams manifest entries via an
// iterator and avoids loading every entry into memory at once.
FetchEntries(fs iceio.IO, discardDeleted bool) ([]ManifestEntry, error)
// contains filtered or unexported methods
}
ManifestFile is the interface for version 1, 2, and 3 manifest files.
func ReadManifestList ¶
func ReadManifestList(in io.Reader) ([]ManifestFile, error)
ReadManifestList reads in an avro manifest list file and returns a slice of manifest files or an error if one is encountered.
Per the Iceberg spec, manifest list files are not required to carry a "format-version" metadata key (only manifest files are). When the key is absent, the version is inferred from the embedded writer schema, so lists from writers that omit the key (e.g. DuckDB's iceberg extension) decode with their real content types and sequence numbers instead of falling back to v1. When the key is present but claims a lower version than the schema carries fields for, an error is returned rather than silently dropping those fields on decode (or on a later rewrite through the older writer schema).
func WriteManifest ¶ added in v0.2.0
func WriteManifest( filename string, out io.Writer, version int, spec PartitionSpec, schema *Schema, snapshotID int64, entries []ManifestEntry, ) (mf ManifestFile, err error)
func WriteManifestV3 ¶ added in v0.7.0
func WriteManifestV3( filename string, out io.Writer, firstRowID int64, spec PartitionSpec, schema *Schema, snapshotID int64, entries []ManifestEntry, ) (mf ManifestFile, nextFirstRowID int64, err error)
WriteManifestV3 writes a v3 data manifest and assigns first_row_id. The returned ManifestFile has FirstRowID() set to firstRowID. nextFirstRowID is firstRowID + AddedRowsCount + ExistingRowsCount for the caller to chain to the next manifest. Use this only for reconstruction — manifests committed through a v3 list writer get first_row_id from AddManifests; pre-setting collides with that allocation.
type ManifestFileOption ¶ added in v0.5.0
type ManifestFileOption func(mf *manifestFile)
func WithManifestFileContent ¶ added in v0.5.0
func WithManifestFileContent(content ManifestContent) ManifestFileOption
WithManifestFileContent sets the ManifestContent of a new manifest file. The default is the content the ManifestWriter was created with, so this is only needed to state that content explicitly. Passing a value that disagrees with the writer makes ToManifestFile fail.
type ManifestListWriter ¶ added in v0.2.0
type ManifestListWriter struct {
// contains filtered or unexported fields
}
func NewManifestListWriterV1 ¶ added in v0.2.0
func NewManifestListWriterV2 ¶ added in v0.2.0
func NewManifestListWriterV3 ¶ added in v0.4.0
func (*ManifestListWriter) AddManifests ¶ added in v0.2.0
func (m *ManifestListWriter) AddManifests(files []ManifestFile) (err error)
AddManifests appends manifest files to the list. If it returns an error, the writer is poisoned: callers must close and discard it, and subsequent calls return the original error.
func (*ManifestListWriter) Close ¶ added in v0.2.0
func (m *ManifestListWriter) Close() error
func (*ManifestListWriter) NextRowID ¶ added in v0.4.0
func (m *ManifestListWriter) NextRowID() *int64
type ManifestReader ¶ added in v0.3.0
type ManifestReader struct {
// contains filtered or unexported fields
}
ManifestReader reads the metadata and data from an avro manifest file. This type is not thread-safe; its methods should not be called from multiple goroutines.
func NewManifestReader ¶ added in v0.3.0
func NewManifestReader(file ManifestFile, in io.Reader) (*ManifestReader, error)
NewManifestReader returns a value that can read the contents of an avro manifest file. If the caller is interested in the manifest entries in the file, it must call [ManifestReader.Entries] before closing the provided reader.
func NewManifestReaderWithProjection ¶ added in v0.7.0
func NewManifestReaderWithProjection( file ManifestFile, in io.Reader, projection ManifestEntryProjection, ) (*ManifestReader, error)
NewManifestReaderWithProjection returns a manifest reader that decodes the standard scan fields and optionally the column statistics selected by projection. Manifest metadata validation and entry inheritance are the same as in NewManifestReader; fields omitted by the projection retain their zero values in the returned DataFile. Entries returned by this reader are read-only planning values and must not be passed to a ManifestWriter, because omitted statistics cannot be recovered on rewrite. ManifestWriter and the DataFile Avro codec reject these values with ErrInvalidArgument.
func (*ManifestReader) Close ¶ added in v0.6.0
func (c *ManifestReader) Close() error
Close releases decoder resources associated with this manifest reader.
func (*ManifestReader) ManifestContent ¶ added in v0.3.0
func (c *ManifestReader) ManifestContent() ManifestContent
ManifestContent returns the type of content in the manifest file.
func (*ManifestReader) PartitionSpec ¶ added in v0.3.0
func (c *ManifestReader) PartitionSpec() (*PartitionSpec, error)
PartitionSpec returns the partition spec encoded in the avro file's metadata.
func (*ManifestReader) PartitionSpecID ¶ added in v0.3.0
func (c *ManifestReader) PartitionSpecID() (int, error)
PartitionSpecID returns the partition spec ID encoded in the avro file's metadata.
func (*ManifestReader) ReadEntry ¶ added in v0.3.0
func (c *ManifestReader) ReadEntry() (ManifestEntry, error)
ReadEntry reads the next manifest entry in the avro file's data.
func (*ManifestReader) Schema ¶ added in v0.3.0
func (c *ManifestReader) Schema() (*Schema, error)
Schema returns the schema encoded in the avro file's metadata.
func (*ManifestReader) SchemaID ¶ added in v0.3.0
func (c *ManifestReader) SchemaID() (int, error)
SchemaID returns the schema ID encoded in the avro file's metadata.
func (*ManifestReader) Version ¶ added in v0.3.0
func (c *ManifestReader) Version() int
Version returns the file's format version.
type ManifestWriter ¶ added in v0.2.0
type ManifestWriter struct {
// contains filtered or unexported fields
}
func NewManifestWriter ¶ added in v0.2.0
func NewManifestWriter(version int, out io.Writer, spec PartitionSpec, schema *Schema, snapshotID int64, opts ...ManifestWriterOption) (*ManifestWriter, error)
func (*ManifestWriter) Add ¶ added in v0.2.0
func (w *ManifestWriter) Add(entry ManifestEntry) error
func (*ManifestWriter) Close ¶ added in v0.2.0
func (w *ManifestWriter) Close() error
func (*ManifestWriter) Delete ¶ added in v0.2.0
func (w *ManifestWriter) Delete(entry ManifestEntry) error
func (*ManifestWriter) Existing ¶ added in v0.2.0
func (w *ManifestWriter) Existing(entry ManifestEntry) error
func (*ManifestWriter) ToManifestFile ¶ added in v0.2.0
func (w *ManifestWriter) ToManifestFile(location string, length int64, opts ...ManifestFileOption) (ManifestFile, error)
type ManifestWriterOption ¶ added in v0.5.0
type ManifestWriterOption func(w *ManifestWriter)
func WithManifestWriterContent ¶ added in v0.5.0
func WithManifestWriterContent(content ManifestContent) ManifestWriterOption
type MapType ¶
type MapType struct {
KeyID int `json:"key-id"`
KeyType Type `json:"-"`
ValueID int `json:"value-id"`
ValueType Type `json:"-"`
ValueRequired bool `json:"value-required"`
}
func (*MapType) Fields ¶
func (m *MapType) Fields() []NestedField
func (*MapType) KeyField ¶
func (m *MapType) KeyField() NestedField
func (*MapType) MarshalJSON ¶
func (*MapType) UnmarshalJSON ¶
func (*MapType) ValueField ¶
func (m *MapType) ValueField() NestedField
type MappedField ¶ added in v0.2.0
type MappedField struct {
Names []string `json:"names"`
// iceberg spec says this is optional, but I don't see any examples
// of this being left empty. Does pyiceberg need to be updated or should
// the spec not say field-id is optional?
FieldID *int `json:"field-id,omitempty"`
Fields []MappedField `json:"fields,omitempty"`
}
func (*MappedField) GetField ¶ added in v0.2.0
func (m *MappedField) GetField(field string) *MappedField
func (*MappedField) ID ¶ added in v0.2.0
func (m *MappedField) ID() int
func (*MappedField) Len ¶ added in v0.2.0
func (m *MappedField) Len() int
func (*MappedField) String ¶ added in v0.2.0
func (m *MappedField) String() string
type MonthTransform ¶
type MonthTransform struct{}
MonthTransform transforms a datetime value into a month value.
func (MonthTransform) Apply ¶
func (MonthTransform) Apply(value Optional[Literal]) (out Optional[Literal])
func (MonthTransform) CanTransform ¶ added in v0.4.0
func (t MonthTransform) CanTransform(sourceType Type) bool
func (MonthTransform) Equals ¶
func (MonthTransform) Equals(other Transform) bool
func (MonthTransform) MarshalText ¶
func (t MonthTransform) MarshalText() ([]byte, error)
func (MonthTransform) PreservesOrder ¶ added in v0.2.0
func (MonthTransform) PreservesOrder() bool
func (MonthTransform) Project ¶
func (t MonthTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (MonthTransform) ResultType ¶
func (MonthTransform) ResultType(Type) Type
func (MonthTransform) String ¶
func (MonthTransform) String() string
func (MonthTransform) ToHumanStr ¶ added in v0.2.0
func (t MonthTransform) ToHumanStr(val any) string
func (MonthTransform) ToHumanStrType ¶ added in v0.7.0
func (t MonthTransform) ToHumanStrType(_ Type, val any) string
func (MonthTransform) Transformer ¶
type NameMapping ¶ added in v0.2.0
type NameMapping []MappedField
func UpdateNameMapping ¶ added in v0.4.0
func UpdateNameMapping(nameMapping NameMapping, updates map[int]NestedField, adds map[int][]NestedField) (NameMapping, error)
UpdateNameMapping performs incremental updates to an existing NameMapping, preserving backward compatibility by maintaining existing field name mappings while adding new ones. This is different from createMappingFromSchema which creates a completely new mapping and loses all historical field name mappings.
For example, when updating a field name:
Original: {FieldID: 1, Names: ["foo"]}
After update: {FieldID: 1, Names: ["foo", "foo_update"]}
This preserves compatibility with existing data files that reference the old field names.
func (NameMapping) String ¶ added in v0.2.0
func (nm NameMapping) String() string
type NameMappingAccessor ¶ added in v0.2.0
type NameMappingAccessor struct{}
func (NameMappingAccessor) FieldPartner ¶ added in v0.2.0
func (NameMappingAccessor) FieldPartner(partnerStruct *MappedField, _ int, fieldName string) *MappedField
func (NameMappingAccessor) ListElementPartner ¶ added in v0.2.0
func (NameMappingAccessor) ListElementPartner(partnerList *MappedField) *MappedField
func (NameMappingAccessor) MapKeyPartner ¶ added in v0.2.0
func (NameMappingAccessor) MapKeyPartner(partnerMap *MappedField) *MappedField
func (NameMappingAccessor) MapValuePartner ¶ added in v0.2.0
func (NameMappingAccessor) MapValuePartner(partnerMap *MappedField) *MappedField
func (NameMappingAccessor) SchemaPartner ¶ added in v0.2.0
func (NameMappingAccessor) SchemaPartner(partner *MappedField) *MappedField
type NameMappingVisitor ¶ added in v0.2.0
type NameMappingVisitor[S, T any] interface { Mapping(nm NameMapping, fieldResults S) S Fields(st []MappedField, fieldResults []T) S Field(field MappedField, fieldResult S) T }
type NestedField ¶
type NestedField struct {
Type `json:"-"`
ID int `json:"id"`
Name string `json:"name"`
Required bool `json:"required"`
Doc string `json:"doc,omitempty"`
InitialDefault any `json:"initial-default,omitempty"`
WriteDefault any `json:"write-default,omitempty"`
}
func LastUpdatedSequenceNumber ¶ added in v0.6.0
func LastUpdatedSequenceNumber() NestedField
LastUpdatedSequenceNumber returns a NestedField for _last_updated_sequence_number (optional long).
func RowID ¶ added in v0.6.0
func RowID() NestedField
RowID returns a NestedField for _row_id (optional long) for use in schemas that include row lineage.
func (*NestedField) Equals ¶
func (n *NestedField) Equals(other NestedField) bool
func (NestedField) MarshalJSON ¶
func (n NestedField) MarshalJSON() ([]byte, error)
func (NestedField) String ¶
func (n NestedField) String() string
func (*NestedField) UnmarshalJSON ¶
func (n *NestedField) UnmarshalJSON(b []byte) error
type NestedType ¶
type NestedType interface {
Type
Fields() []NestedField
}
NestedType is an interface that allows access to the child fields of a nested type such as a list/struct/map type.
type NotExpr ¶
type NotExpr struct {
// contains filtered or unexported fields
}
func (NotExpr) Equals ¶
func (n NotExpr) Equals(other BooleanExpression) bool
func (NotExpr) MarshalJSON ¶ added in v0.7.0
func (NotExpr) Negate ¶
func (n NotExpr) Negate() BooleanExpression
type NumericLiteral ¶
type Operation ¶
type Operation int
Operation is an enum used for constants to define what operation a given expression or predicate is going to execute.
const ( OpTrue Operation = iota // True OpFalse // False // unary ops OpIsNull // IsNull OpNotNull // NotNull OpIsNan // IsNaN OpNotNan // NotNaN // literal ops OpLT // LessThan OpLTEQ // LessThanEqual OpGT // GreaterThan OpGTEQ // GreaterThanEqual OpEQ // Equal OpNEQ // NotEqual OpStartsWith // StartsWith OpNotStartsWith // NotStartsWith // set ops OpIn // In OpNotIn // NotIn // boolean ops OpNot // Not OpAnd // And OpOr // Or // geospatial ops. Kept after the boolean ops so the group ranges above // (used for quick op-kind validation) are undisturbed. These have their own // predicate constructor (BBoxIntersects) rather than belonging to the // unary/literal/set groups. OpBBoxIntersects // BBoxIntersects OpBBoxNotIntersects // BBoxNotIntersects )
func (Operation) FlipLR ¶
FlipLR returns the correct operation to use if the left and right operands are flipped.
type OrExpr ¶
type OrExpr struct {
// contains filtered or unexported fields
}
func (OrExpr) Equals ¶
func (o OrExpr) Equals(other BooleanExpression) bool
func (OrExpr) MarshalJSON ¶ added in v0.7.0
func (OrExpr) Negate ¶
func (o OrExpr) Negate() BooleanExpression
type PartitionField ¶
type PartitionField struct {
// SourceIDs contains the source column ids from the table's schema.
// For single-argument transforms this will have exactly one element.
// For multi-argument transforms this will have multiple elements.
SourceIDs []int `json:"-"`
// FieldID is the partition field id across all the table partition specs
FieldID int `json:"field-id"`
// Name is the name of the partition field itself
Name string `json:"name"`
// Transform is the transform used to produce the partition value
Transform Transform `json:"transform"`
// contains filtered or unexported fields
}
PartitionField represents how one partition value is derived from the source column by transformation.
func (PartitionField) Equals ¶ added in v0.5.0
func (p PartitionField) Equals(other PartitionField) bool
func (*PartitionField) EscapedName ¶ added in v0.5.0
func (p *PartitionField) EscapedName() string
EscapedName returns the URL-escaped version of the partition field name. initialize() pre-populates escapedName for specs built through a constructor.
func (PartitionField) MarshalJSON ¶ added in v0.6.0
func (p PartitionField) MarshalJSON() ([]byte, error)
func (PartitionField) SourceID ¶
func (p PartitionField) SourceID() int
SourceID returns the first source column id. For single-argument transforms this is the only source column. For multi-argument transforms this is the first source column.
func (*PartitionField) String ¶
func (p *PartitionField) String() string
func (*PartitionField) UnmarshalJSON ¶
func (p *PartitionField) UnmarshalJSON(b []byte) error
type PartitionOption ¶ added in v0.4.0
type PartitionOption func(*PartitionSpec) error
func AddPartitionFieldByName ¶ added in v0.4.0
func AddPartitionFieldBySourceID ¶ added in v0.4.0
func WithSpecID ¶ added in v0.4.0
func WithSpecID(id int) PartitionOption
type PartitionSpec ¶
type PartitionSpec struct {
// contains filtered or unexported fields
}
PartitionSpec captures the transformation from table data to partition values
func NewPartitionSpec ¶
func NewPartitionSpec(fields ...PartitionField) PartitionSpec
NewPartitionSpec creates a new PartitionSpec with the given fields.
The fields are not verified against a schema, use NewPartitionSpecOpts if you have to ensure compatibility.
The fields are not checked for redundancy either, so this accepts a spec that UnmarshalJSON would reject, meaning the result may not survive a metadata round trip. Use NewPartitionSpecOpts when the spec has to be readable back.
func NewPartitionSpecID ¶
func NewPartitionSpecID(id int, fields ...PartitionField) PartitionSpec
NewPartitionSpecID creates a new PartitionSpec with the given fields and id.
The fields are not verified against a schema, use NewPartitionSpecOpts if you have to ensure compatibility.
The fields are not checked for redundancy either, so this accepts a spec that UnmarshalJSON would reject, meaning the result may not survive a metadata round trip. Use NewPartitionSpecOpts when the spec has to be readable back.
func NewPartitionSpecOpts ¶ added in v0.4.0
func NewPartitionSpecOpts(opts ...PartitionOption) (PartitionSpec, error)
NewPartitionSpecOpts assembles a brand new spec and validates it as authored, so it permits at most one time transform per source column. That rule is narrower than the replay rule, so anything it accepts still parses back.
func (*PartitionSpec) BindToSchema ¶ added in v0.4.0
func (p *PartitionSpec) BindToSchema(schema *Schema, lastPartitionID *int, newSpecID *int) (PartitionSpec, error)
BindToSchema creates a new PartitionSpec by copying the fields from the existing spec verifying compatibility with the schema.
If newSpecID is not nil, it will be used as the spec id for the new spec. Otherwise, the existing spec id will be used. If a field in the spec is incompatible with the schema, an error will be returned.
func (*PartitionSpec) Clone ¶ added in v0.7.0
func (ps *PartitionSpec) Clone() PartitionSpec
Clone returns a deep copy of the partition spec, including mutable transform values and the source-ID lookup index.
func (*PartitionSpec) CompatibleWith ¶
func (ps *PartitionSpec) CompatibleWith(other *PartitionSpec) bool
CompatibleWith returns true if this partition spec is considered compatible with the passed in partition spec. This means that the two specs have equivalent field lists regardless of the spec id.
func (PartitionSpec) Equals ¶
func (ps PartitionSpec) Equals(other PartitionSpec) bool
Equals returns true iff the field lists are the same AND the spec id is the same between this partition spec and the provided one.
func (*PartitionSpec) Field ¶
func (ps *PartitionSpec) Field(i int) PartitionField
func (*PartitionSpec) Fields ¶ added in v0.2.0
func (ps *PartitionSpec) Fields() iter.Seq2[int, PartitionField]
Fields returns an iterator over the partition fields in this spec.
func (*PartitionSpec) FieldsBySourceID ¶
func (ps *PartitionSpec) FieldsBySourceID(fieldID int) []PartitionField
func (*PartitionSpec) FieldsRef ¶ added in v0.7.0
func (ps *PartitionSpec) FieldsRef(_ internal.PartitionSpecRef) []PartitionField
FieldsRef returns the partition fields owned by this spec for trusted internal callers. The returned slice and everything reachable through the fields must be treated as read-only.
func (*PartitionSpec) ID ¶
func (ps *PartitionSpec) ID() int
func (PartitionSpec) IsUnpartitioned ¶
func (ps PartitionSpec) IsUnpartitioned() bool
func (*PartitionSpec) LastAssignedFieldID ¶
func (ps *PartitionSpec) LastAssignedFieldID() int
func (*PartitionSpec) Len ¶ added in v0.4.0
func (p *PartitionSpec) Len() int
func (PartitionSpec) MarshalJSON ¶
func (ps PartitionSpec) MarshalJSON() ([]byte, error)
func (*PartitionSpec) NumFields ¶
func (ps *PartitionSpec) NumFields() int
func (*PartitionSpec) PartitionToPath ¶ added in v0.2.0
func (ps *PartitionSpec) PartitionToPath(data StructLike, sc *Schema) string
PartitionToPath produces a proper partition path from the data and schema by converting the values to human readable strings and properly escaping.
The path will be in the form of `name1=value1/name2=value2/...`.
This does not apply the transforms to the data, it is assumed the provided data has already been transformed appropriately.
func (*PartitionSpec) PartitionType ¶
func (ps *PartitionSpec) PartitionType(schema *Schema) *StructType
PartitionType produces a struct of the partition spec.
The partition fields should be optional:
- All partition transforms are required to produce null if the input value is null. This can happen when the source column is optional.
- Partition fields may be added later, in which case not all files would have the result field and it may be null.
There is a case where we can guarantee that a partition field in the first and only partition spec that uses a required source column will never be null, but it doesn't seem worth tracking this case.
If a source column is missing, UnknownType is passed to the transform. This retains the field's position and lets transforms with fixed result types, such as bucket, continue to resolve their result type.
func (PartitionSpec) String ¶
func (ps PartitionSpec) String() string
func (*PartitionSpec) UnmarshalJSON ¶
func (ps *PartitionSpec) UnmarshalJSON(b []byte) error
type PartnerAccessor ¶
type PreOrderSchemaVisitor ¶ added in v0.2.0
type PreOrderSchemaVisitor[T any] interface { Schema(*Schema, func() T) T Struct(StructType, []func() T) T Field(NestedField, func() T) T List(ListType, func() T) T Map(MapType, func() T, func() T) T Primitive(PrimitiveType) T Variant(VariantType) T }
type PrimitiveType ¶
type PrimitiveType interface {
Type
// contains filtered or unexported methods
}
type Properties ¶
func (Properties) Get ¶
func (p Properties) Get(key, defVal string) string
Get returns the value of the key if it exists, otherwise it returns the default value.
func (Properties) GetBool ¶ added in v0.2.0
func (p Properties) GetBool(key string, defVal bool) bool
func (Properties) GetInt64 ¶ added in v0.7.0
func (p Properties) GetInt64(key string, defVal int64) int64
GetInt64 reads a 64-bit integer property by key. A missing key or an unparseable value returns defVal. Unlike GetInt, this avoids truncating large int64 sentinel values (such as math.MaxInt64) on 32-bit platforms where int is only 32 bits wide.
func (Properties) GetUInt64 ¶ added in v0.7.0
func (p Properties) GetUInt64(key string, defVal uint64) uint64
GetUInt64 reads an unsigned-integer property by key. A missing key, an unparseable value, or a negative value returns defVal. GetUInt64 uses strconv.ParseUint, which rejects negatives rather than silently wrapping them to a large positive number.
type Reference ¶
type Reference string
Reference is a field name not yet bound to a particular field in a schema
func (Reference) Equals ¶
func (r Reference) Equals(other UnboundTerm) bool
func (Reference) MarshalJSON ¶ added in v0.7.0
type Schema ¶
type Schema struct {
ID int `json:"schema-id"`
IdentifierFieldIDs []int `json:"identifier-field-ids"`
// contains filtered or unexported fields
}
Schema is an Iceberg table schema, represented as a struct with multiple fields. The fields are only exported via accessor methods rather than exposing the slice directly in order to ensure a schema as immutable.
func ApplyNameMapping ¶ added in v0.2.0
func ApplyNameMapping(schemaWithoutIDs *Schema, nameMapping NameMapping) (*Schema, error)
func AssignFreshSchemaIDs ¶ added in v0.2.0
AssignFreshSchemaIDs creates a new schema with fresh field IDs for all of the fields in it. The nextID function is used to iteratively generate the ids, if it is nil then a simple incrementing counter is used starting at 1.
func NewSchema ¶
func NewSchema(id int, fields ...NestedField) *Schema
NewSchema constructs a new schema with the provided ID and list of fields.
func NewSchemaFromJsonFields ¶ added in v0.3.0
NewSchemaFromJsonFields constructs a new schema with the provided ID and fields in json form
func NewSchemaWithIdentifiers ¶
func NewSchemaWithIdentifiers(id int, identifierIDs []int, fields ...NestedField) *Schema
NewSchemaWithIdentifiers constructs a new schema with the provided ID and fields, along with a slice of field IDs to be listed as identifier fields.
func PruneColumns ¶
PruneColumns visits a schema pruning any columns which do not exist in the provided selected set. Parent fields of a selected child will be retained.
func SanitizeColumnNames ¶ added in v0.3.0
SanitizeColumnNames returns a copy of sc whose field names are compatible with Java Iceberg's Avro name sanitization. Characters outside the BMP are escaped as UTF-16 surrogate pairs, and only Unicode decimal digits are treated as digits; other numeric categories are escaped.
Empty or invalid UTF-8 field names and names that collide after sanitization return an error wrapping ErrInvalidSchema. Collision errors are reported here before a downstream Avro schema builder encounters the duplicate field name.
func SchemaWithRowID ¶ added in v0.7.0
SchemaWithRowID returns a new schema with only the _row_id metadata column appended. _last_updated_sequence_number is intentionally omitted: leaving it absent in the written Parquet means readers synthesize it from the manifest entry's data_sequence_number, which is the new file's snapshot sequence number after the rewrite — exactly the value the spec requires for rewritten rows without an explicit override.
Idempotent on RowIDFieldID; allocates a fresh field slice.
func SchemaWithRowLineage ¶ added in v0.7.0
SchemaWithRowLineage appends both row-lineage metadata columns (_row_id, _last_updated_sequence_number) so a CoW rewrite or compaction preserves them.
func SchemaWithRowLineageColumns ¶ added in v0.7.0
SchemaWithRowLineageColumns appends the requested row-lineage columns: _row_id when rowID, _last_updated_sequence_number when lastUpdatedSeq. Request exactly the columns you will materialize so the read schema matches the produced batch.
Idempotent by reserved field ID; always clones the field slice (no aliasing).
func (*Schema) AsStruct ¶
func (s *Schema) AsStruct() StructType
AsStruct returns a Struct with the same fields as the schema which can then be used as a Type.
func (*Schema) Equals ¶
Equals compares the fields and identifierIDs, but does not compare the schema ID itself.
func (*Schema) Field ¶
func (s *Schema) Field(i int) NestedField
func (*Schema) FieldHasOptionalParent ¶
func (*Schema) Fields ¶
func (s *Schema) Fields() []NestedField
func (*Schema) FieldsRef ¶ added in v0.7.0
func (s *Schema) FieldsRef(_ internal.SchemaRef) []NestedField
FieldsRef returns the schema-owned fields for trusted internal callers. The returned slice and everything reachable through the fields must be treated as read-only.
func (*Schema) FindColumnName ¶
FindColumnName returns the name of the column identified by the passed in field id. The second return value reports whether or not the field id was found in the schema.
func (*Schema) FindFieldByID ¶
func (s *Schema) FindFieldByID(id int) (NestedField, bool)
FindFieldByID is like *Schema.FindColumnName, but returns the whole field rather than just the field name.
func (*Schema) FindFieldByIDRef ¶ added in v0.7.0
FindFieldByIDRef returns a schema-owned field without cloning it. The returned field is a value copy, but its Type and everything reachable from that Type are shared with the schema and must be treated as read-only. This method is limited to internal callers that need to avoid clone-on-read overhead.
func (*Schema) FindFieldByName ¶
func (s *Schema) FindFieldByName(name string) (NestedField, bool)
FindFieldByName returns the field identified by the name given, the second return value will be false if no field by this name is found.
Note: This search is done in a case sensitive manner. To perform a case insensitive search, use *Schema.FindFieldByNameCaseInsensitive.
func (*Schema) FindFieldByNameCaseInsensitive ¶
func (s *Schema) FindFieldByNameCaseInsensitive(name string) (NestedField, bool)
FindFieldByNameCaseInsensitive is like *Schema.FindFieldByName, but performs a case insensitive search.
func (*Schema) FindTypeByID ¶
FindTypeByID is like *Schema.FindFieldByID, but returns only the data type of the field.
func (*Schema) FindTypeByName ¶
FindTypeByName is a convenience function for calling *Schema.FindFieldByName, and then returning just the type.
func (*Schema) FindTypeByNameCaseInsensitive ¶
FindTypeByNameCaseInsensitive is like *Schema.FindTypeByName but performs a case insensitive search.
func (*Schema) FlatFields ¶ added in v0.5.0
func (s *Schema) FlatFields() (iter.Seq[NestedField], error)
FlatFields returns an iterator over the flattened fields in the schema The fields are returned in arbitrary order.
func (*Schema) HighestFieldID ¶
HighestFieldID returns the value of the numerically highest field ID in this schema.
func (*Schema) MarshalJSON ¶
func (*Schema) NameMapping ¶ added in v0.2.0
func (s *Schema) NameMapping() NameMapping
func (*Schema) Select ¶
Select creates a new schema with just the fields identified by name passed in the order they are provided. If caseSensitive is false, then fields will be identified by case insensitive search.
An error is returned if a requested name cannot be found.
func (*Schema) UnmarshalJSON ¶
type SchemaVisitor ¶
type SchemaVisitor[T any] interface { Schema(schema *Schema, structResult T) T Struct(st StructType, fieldResults []T) T Field(field NestedField, fieldResult T) T List(list ListType, elemResult T) T Map(mapType MapType, keyResult, valueResult T) T Primitive(p PrimitiveType) T Variant(v VariantType) T }
SchemaVisitor is an interface that can be implemented to allow for easy traversal and processing of a schema.
A SchemaVisitor can also optionally implement the Before/After Field, ListElement, MapKey, or MapValue interfaces to allow them to get called at the appropriate points within schema traversal.
type SchemaVisitorPerPrimitiveType ¶
type SchemaVisitorPerPrimitiveType[T any] interface { SchemaVisitor[T] VisitFixed(FixedType) T VisitDecimal(DecimalType) T VisitBoolean() T VisitInt32() T VisitInt64() T VisitFloat32() T VisitFloat64() T VisitDate() T VisitTime() T VisitTimestamp() T VisitTimestampNs() T VisitTimestampTz() T VisitTimestampNsTz() T VisitString() T VisitBinary() T VisitUUID() T VisitUnknown() T VisitVariant() T VisitGeometry(GeometryType) T VisitGeography(GeographyType) T }
type SchemaWithPartnerVisitor ¶
type SchemaWithPartnerVisitor[T, P any] interface { Schema(sc *Schema, schemaPartner P, structResult T) T Struct(st StructType, structPartner P, fieldResults []T) T Field(field NestedField, fieldPartner P, fieldResult T) T List(l ListType, listPartner P, elemResult T) T Map(m MapType, mapPartner P, keyResult, valResult T) T Primitive(p PrimitiveType, primitivePartner P) T Variant(v VariantType, variantPartner P) T }
type StringLiteral ¶
type StringLiteral string
func (StringLiteral) Any ¶ added in v0.2.0
func (s StringLiteral) Any() any
func (StringLiteral) Comparator ¶
func (StringLiteral) Comparator() Comparator[string]
func (StringLiteral) Equals ¶
func (s StringLiteral) Equals(other Literal) bool
func (StringLiteral) MarshalBinary ¶
func (s StringLiteral) MarshalBinary() (data []byte, err error)
func (StringLiteral) MarshalJSON ¶ added in v0.7.0
func (l StringLiteral) MarshalJSON() ([]byte, error)
func (StringLiteral) String ¶
func (s StringLiteral) String() string
func (StringLiteral) Type ¶
func (s StringLiteral) Type() Type
func (*StringLiteral) UnmarshalBinary ¶
func (s *StringLiteral) UnmarshalBinary(data []byte) error
func (StringLiteral) Value ¶
func (s StringLiteral) Value() string
type StringType ¶
type StringType struct{}
func (StringType) Equals ¶
func (StringType) Equals(other Type) bool
func (StringType) String ¶
func (StringType) String() string
func (StringType) Type ¶
func (StringType) Type() string
type StructLike ¶ added in v0.6.0
type StructLike interface {
// Size returns the number of columns in this row
Size() int
// Get returns the value in the requested column,
// will panic if pos is out of bounds.
Get(pos int) any
// Set changes the value in the column indicated,
// will panic if pos is out of bounds.
Set(pos int, val any)
}
StructLike represents a single row in a record.
type StructType ¶
type StructType struct {
FieldList []NestedField `json:"fields"`
}
func (*StructType) Equals ¶
func (s *StructType) Equals(other Type) bool
func (*StructType) Fields ¶
func (s *StructType) Fields() []NestedField
func (*StructType) MarshalJSON ¶
func (s *StructType) MarshalJSON() ([]byte, error)
func (*StructType) String ¶
func (s *StructType) String() string
func (*StructType) Type ¶
func (*StructType) Type() string
type TimeLiteral ¶
type TimeLiteral Time
func (TimeLiteral) Any ¶ added in v0.2.0
func (t TimeLiteral) Any() any
func (TimeLiteral) Comparator ¶
func (TimeLiteral) Comparator() Comparator[Time]
func (TimeLiteral) Equals ¶
func (t TimeLiteral) Equals(other Literal) bool
func (TimeLiteral) MarshalBinary ¶
func (t TimeLiteral) MarshalBinary() (data []byte, err error)
func (TimeLiteral) MarshalJSON ¶ added in v0.7.0
func (l TimeLiteral) MarshalJSON() ([]byte, error)
func (TimeLiteral) String ¶
func (t TimeLiteral) String() string
func (TimeLiteral) Type ¶
func (t TimeLiteral) Type() Type
func (*TimeLiteral) UnmarshalBinary ¶
func (t *TimeLiteral) UnmarshalBinary(data []byte) error
func (TimeLiteral) Value ¶
func (t TimeLiteral) Value() Time
type TimeTransform ¶ added in v0.4.0
type Timestamp ¶
type Timestamp int64
func (Timestamp) ToNanos ¶ added in v0.5.0
func (t Timestamp) ToNanos() TimestampNano
type TimestampLiteral ¶
type TimestampLiteral Timestamp
func (TimestampLiteral) Any ¶ added in v0.2.0
func (t TimestampLiteral) Any() any
func (TimestampLiteral) Comparator ¶
func (TimestampLiteral) Comparator() Comparator[Timestamp]
func (TimestampLiteral) Decrement ¶
func (t TimestampLiteral) Decrement() Literal
func (TimestampLiteral) Equals ¶
func (t TimestampLiteral) Equals(other Literal) bool
func (TimestampLiteral) Increment ¶
func (t TimestampLiteral) Increment() Literal
func (TimestampLiteral) MarshalBinary ¶
func (t TimestampLiteral) MarshalBinary() (data []byte, err error)
func (TimestampLiteral) MarshalJSON ¶ added in v0.7.0
func (TimestampLiteral) MarshalJSON() ([]byte, error)
A bare timestamp literal has no field type, so it can't pick a wire form; serialize through a predicate (see literalValue) instead.
func (TimestampLiteral) String ¶
func (t TimestampLiteral) String() string
func (TimestampLiteral) Type ¶
func (t TimestampLiteral) Type() Type
func (*TimestampLiteral) UnmarshalBinary ¶
func (t *TimestampLiteral) UnmarshalBinary(data []byte) error
func (TimestampLiteral) Value ¶
func (t TimestampLiteral) Value() Timestamp
type TimestampNano ¶ added in v0.5.0
type TimestampNano int64
func (TimestampNano) ToDate ¶ added in v0.5.0
func (t TimestampNano) ToDate() Date
func (TimestampNano) ToMicros ¶ added in v0.5.0
func (t TimestampNano) ToMicros() Timestamp
func (TimestampNano) ToTime ¶ added in v0.5.0
func (t TimestampNano) ToTime() time.Time
type TimestampNsLiteral ¶ added in v0.5.0
type TimestampNsLiteral TimestampNano
func (TimestampNsLiteral) Any ¶ added in v0.5.0
func (t TimestampNsLiteral) Any() any
func (TimestampNsLiteral) Comparator ¶ added in v0.5.0
func (TimestampNsLiteral) Comparator() Comparator[TimestampNano]
func (TimestampNsLiteral) Decrement ¶ added in v0.5.0
func (t TimestampNsLiteral) Decrement() Literal
func (TimestampNsLiteral) Equals ¶ added in v0.5.0
func (t TimestampNsLiteral) Equals(other Literal) bool
func (TimestampNsLiteral) Increment ¶ added in v0.5.0
func (t TimestampNsLiteral) Increment() Literal
func (TimestampNsLiteral) MarshalBinary ¶ added in v0.5.0
func (t TimestampNsLiteral) MarshalBinary() (data []byte, err error)
func (TimestampNsLiteral) MarshalJSON ¶ added in v0.7.0
func (TimestampNsLiteral) MarshalJSON() ([]byte, error)
func (TimestampNsLiteral) String ¶ added in v0.5.0
func (t TimestampNsLiteral) String() string
func (TimestampNsLiteral) To ¶ added in v0.5.0
func (t TimestampNsLiteral) To(typ Type) (Literal, error)
func (TimestampNsLiteral) Type ¶ added in v0.5.0
func (t TimestampNsLiteral) Type() Type
func (*TimestampNsLiteral) UnmarshalBinary ¶ added in v0.5.0
func (t *TimestampNsLiteral) UnmarshalBinary(data []byte) error
func (TimestampNsLiteral) Value ¶ added in v0.5.0
func (t TimestampNsLiteral) Value() TimestampNano
type TimestampNsType ¶ added in v0.5.0
type TimestampNsType struct{}
TimestampNsType represents a timestamp stored as nanoseconds since the unix epoch without regard for timezone. Requires format version 3+.
func (TimestampNsType) Equals ¶ added in v0.5.0
func (TimestampNsType) Equals(other Type) bool
func (TimestampNsType) String ¶ added in v0.5.0
func (TimestampNsType) String() string
func (TimestampNsType) Type ¶ added in v0.5.0
func (TimestampNsType) Type() string
type TimestampType ¶
type TimestampType struct{}
TimestampType represents a number of microseconds since the unix epoch without regard for timezone.
func (TimestampType) Equals ¶
func (TimestampType) Equals(other Type) bool
func (TimestampType) String ¶
func (TimestampType) String() string
func (TimestampType) Type ¶
func (TimestampType) Type() string
type TimestampTzNsType ¶ added in v0.5.0
type TimestampTzNsType struct{}
TimestampTzNsType represents a timestamp stored as UTC with nanoseconds since the unix epoch. Requires format version 3+.
func (TimestampTzNsType) Equals ¶ added in v0.5.0
func (TimestampTzNsType) Equals(other Type) bool
func (TimestampTzNsType) String ¶ added in v0.5.0
func (TimestampTzNsType) String() string
func (TimestampTzNsType) Type ¶ added in v0.5.0
func (TimestampTzNsType) Type() string
type TimestampTzType ¶
type TimestampTzType struct{}
TimestampTzType represents a timestamp stored as UTC representing the number of microseconds since the unix epoch.
func (TimestampTzType) Equals ¶
func (TimestampTzType) Equals(other Type) bool
func (TimestampTzType) String ¶
func (TimestampTzType) String() string
func (TimestampTzType) Type ¶
func (TimestampTzType) Type() string
type Transform ¶
type Transform interface {
fmt.Stringer
encoding.TextMarshaler
CanTransform(t Type) bool
ResultType(t Type) Type
PreservesOrder() bool
Equals(Transform) bool
Apply(Optional[Literal]) Optional[Literal]
Project(name string, pred BoundPredicate) (UnboundPredicate, error)
ToHumanStrType(typ Type, val any) string
// Deprecated: ToHumanStr cannot recover source-type information; use ToHumanStrType instead.
ToHumanStr(any) string
}
Transform is an interface for the various Transformation types in partition specs. Currently, they do not yet provide actual transformation functions or implementation. That will come later as data reading gets implemented.
func ParseTransform ¶
ParseTransform takes the string representation of a transform as defined in the iceberg spec, and produces the appropriate Transform object. Strings that don't name a known transform yield an UnknownTransform rather than an error, so that v3 tables using transforms this implementation doesn't recognize can still be read. This mirrors Java's Transforms.fromString: only a bucket/truncate whose width parses as a number but is out of range (e.g. bucket[0]) is an error; anything else that doesn't match the width pattern, like bucket[-1], is an unknown transform.
type TruncateTransform ¶
type TruncateTransform struct {
Width int
}
TruncateTransform is a transformation for truncating a value to a specified width.
func (TruncateTransform) Apply ¶
func (t TruncateTransform) Apply(value Optional[Literal]) (out Optional[Literal])
func (TruncateTransform) CanTransform ¶ added in v0.4.0
func (TruncateTransform) CanTransform(t Type) bool
func (TruncateTransform) Equals ¶
func (t TruncateTransform) Equals(other Transform) bool
func (TruncateTransform) MarshalText ¶
func (t TruncateTransform) MarshalText() ([]byte, error)
func (TruncateTransform) PreservesOrder ¶ added in v0.2.0
func (TruncateTransform) PreservesOrder() bool
func (TruncateTransform) Project ¶
func (t TruncateTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (TruncateTransform) ResultType ¶
func (TruncateTransform) ResultType(t Type) Type
func (TruncateTransform) String ¶
func (t TruncateTransform) String() string
func (TruncateTransform) ToHumanStr ¶ added in v0.2.0
func (TruncateTransform) ToHumanStr(val any) string
func (TruncateTransform) ToHumanStrType ¶ added in v0.7.0
func (t TruncateTransform) ToHumanStrType(_ Type, val any) string
func (TruncateTransform) Transformer ¶
func (t TruncateTransform) Transformer(src Type) (func(any) any, error)
type Type ¶
Type is an interface representing any of the available iceberg types, such as primitives (int32/int64/etc.) or nested types (list/struct/map).
func PromoteType ¶
PromoteType promotes the type being read from a file to a requested read type. fileType is the type from the file being read readType is the requested readType
type TypedLiteral ¶
type TypedLiteral[T LiteralType] interface { Literal Value() T Comparator() Comparator[T] }
TypedLiteral is a generic interface for Literals so that you can retrieve the value. This is based on the physical representative type, which means that FixedLiteral and BinaryLiteral will both return []byte, etc.
type UUIDLiteral ¶
func (UUIDLiteral) Any ¶ added in v0.2.0
func (u UUIDLiteral) Any() any
func (UUIDLiteral) Comparator ¶
func (UUIDLiteral) Comparator() Comparator[uuid.UUID]
func (UUIDLiteral) Equals ¶
func (u UUIDLiteral) Equals(other Literal) bool
func (UUIDLiteral) MarshalBinary ¶
func (u UUIDLiteral) MarshalBinary() (data []byte, err error)
func (UUIDLiteral) MarshalJSON ¶ added in v0.7.0
func (l UUIDLiteral) MarshalJSON() ([]byte, error)
func (UUIDLiteral) String ¶
func (u UUIDLiteral) String() string
func (UUIDLiteral) Type ¶
func (UUIDLiteral) Type() Type
func (*UUIDLiteral) UnmarshalBinary ¶
func (u *UUIDLiteral) UnmarshalBinary(data []byte) error
func (UUIDLiteral) Value ¶
func (u UUIDLiteral) Value() uuid.UUID
type UnboundPartitionSpec ¶ added in v0.7.0
type UnboundPartitionSpec struct {
PartitionSpec
}
UnboundPartitionSpec decodes a partition spec whose source IDs have not yet been resolved against the schema that decides them, so unlike bound field IDs they need not be positive. Two wire forms arrive this way:
- A create-table request's spec, which carries the client's placeholder source IDs rather than table field IDs. A client numbers those placeholders however it likes: Spark numbers the columns of a new table from zero, so partitioning by the first column arrives as source-id 0.
- An add-spec commit payload, whose source IDs are the current schema's field IDs, except that a dropped partition field arrives as a void transform over source-id 0. Applying the update binds the spec to the table's current schema, which is what decides whether the rest resolve.
A create-table spec's placeholders are field IDs of the schema the client sent with it, so BindToSchema resolves them only when given that same schema. Passing the schema the table ends up with instead binds against unrelated IDs, and a placeholder that collides with one of them binds silently to the wrong column. table.NewMetadata does the whole create-table flow, resolving the placeholders through the request schema by name and reassigning fresh IDs.
Use PartitionSpec for specs read from table metadata, where source IDs are bound field IDs: positive, apart from the void tombstone a dropped field leaves behind. Catalog implementations should decode both the REST create-table request's spec and an add-spec commit payload into this type.
func (*UnboundPartitionSpec) UnmarshalJSON ¶ added in v0.7.0
func (u *UnboundPartitionSpec) UnmarshalJSON(b []byte) error
type UnboundPredicate ¶
type UnboundPredicate interface {
BooleanExpression
Term() UnboundTerm
// contains filtered or unexported methods
}
An UnboundPredicate represents a boolean predicate expression which has not yet been bound to a schema. Binding it will produce a BooleanExpression.
BooleanExpression is used for the binding result because we may optimize and return AlwaysTrue / AlwaysFalse in some scenarios during binding which are not considered to be "Bound" as they do not have a bound Term or Reference.
func BBoxIntersects ¶ added in v0.7.0
func BBoxIntersects(t UnboundTerm, bbox BoundingBox) UnboundPredicate
BBoxIntersects constructs an unbound geospatial predicate that matches rows whose geometry/geography value's bounding box intersects the query box. It is used to prune data files whose stored geo bounds cannot overlap the query region; the spec only requires bbox-based pruning, so full geometric predicate evaluation (ST_Intersects/ST_Within) remains a query-engine concern.
Panics if the term is nil or the bbox is not Valid (NaN coordinate or an inverted min/max), either of which would cause silent mis-pruning. Panicking on invalid construction is consistent with the other predicate constructors in this package (UnaryPredicate, LiteralPredicate, SetPredicate, NewAnd, NewNot), which all panic rather than return an error; a caller building a box from untrusted input should validate it with BoundingBox.Valid first.
func EqualTo ¶
func EqualTo[T LiteralType](t UnboundTerm, v T) UnboundPredicate
EqualTo is a convenience wrapper for calling LiteralPredicate(OpEQ, t, NewLiteral(v))
Will panic if t is nil
func GreaterThan ¶
func GreaterThan[T LiteralType](t UnboundTerm, v T) UnboundPredicate
GreaterThan is a convenience wrapper for calling LiteralPredicate(OpGT, t, NewLiteral(v))
Will panic if t is nil
func GreaterThanEqual ¶
func GreaterThanEqual[T LiteralType](t UnboundTerm, v T) UnboundPredicate
GreaterThanEqual is a convenience wrapper for calling LiteralPredicate(OpGTEQ, t, NewLiteral(v))
Will panic if t is nil
func IsNaN ¶
func IsNaN(t UnboundTerm) UnboundPredicate
IsNaN is a convenience wrapper for calling UnaryPredicate(OpIsNan, t)
Will panic if t is nil
func IsNull ¶
func IsNull(t UnboundTerm) UnboundPredicate
IsNull is a convenience wrapper for calling UnaryPredicate(OpIsNull, t)
Will panic if t is nil
func LessThan ¶
func LessThan[T LiteralType](t UnboundTerm, v T) UnboundPredicate
LessThan is a convenience wrapper for calling LiteralPredicate(OpLT, t, NewLiteral(v))
Will panic if t is nil
func LessThanEqual ¶
func LessThanEqual[T LiteralType](t UnboundTerm, v T) UnboundPredicate
LessThanEqual is a convenience wrapper for calling LiteralPredicate(OpLTEQ, t, NewLiteral(v))
Will panic if t is nil
func LiteralPredicate ¶
func LiteralPredicate(op Operation, t UnboundTerm, lit Literal) UnboundPredicate
LiteralPredicate constructs an unbound predicate for an operation that requires a single literal argument, such as LessThan or StartsWith.
Panics if the operation provided is not a valid Literal operation, if the term is nil or if the literal is nil.
func NotEqualTo ¶
func NotEqualTo[T LiteralType](t UnboundTerm, v T) UnboundPredicate
NotEqualTo is a convenience wrapper for calling LiteralPredicate(OpNEQ, t, NewLiteral(v))
Will panic if t is nil
func NotNaN ¶
func NotNaN(t UnboundTerm) UnboundPredicate
NotNaN is a convenience wrapper for calling UnaryPredicate(OpNotNan, t)
Will panic if t is nil
func NotNull ¶
func NotNull(t UnboundTerm) UnboundPredicate
NotNull is a convenience wrapper for calling UnaryPredicate(OpNotNull, t)
Will panic if t is nil
func NotStartsWith ¶
func NotStartsWith(t UnboundTerm, v string) UnboundPredicate
NotStartsWith is a convenience wrapper for calling LiteralPredicate(OpNotStartsWith, t, NewLiteral(v))
Will panic if t is nil
func StartsWith ¶
func StartsWith(t UnboundTerm, v string) UnboundPredicate
StartsWith is a convenience wrapper for calling LiteralPredicate(OpStartsWith, t, NewLiteral(v))
Will panic if t is nil
func UnaryPredicate ¶
func UnaryPredicate(op Operation, t UnboundTerm) UnboundPredicate
UnaryPredicate creates and returns an unbound predicate for the provided unary operation. Will panic if op is not a unary operation.
type UnboundTerm ¶
type UnboundTerm interface {
Term
Equals(UnboundTerm) bool
Bind(schema *Schema, caseSensitive bool) (BoundTerm, error)
}
UnboundTerm is an expression that evaluates to a value that isn't yet bound to a schema, thus it isn't yet known what the type will be.
func Extract ¶ added in v0.7.0
func Extract(ref Reference, path string, typ PrimitiveType) UnboundTerm
Extract creates an unbound variant sub-path term for a dotted JSONPath.
type UnboundTransform ¶ added in v0.7.0
type UnboundTransform struct {
// contains filtered or unexported fields
}
UnboundTransform is a transform applied to a term, not yet bound to a schema; the unbound counterpart of BoundTransform. It's what a transform term in a REST expression (e.g. bucket[16](id)) parses into.
func NewUnboundTransform ¶ added in v0.7.0
func NewUnboundTransform(transform Transform, term UnboundTerm) *UnboundTransform
func (*UnboundTransform) Bind ¶ added in v0.7.0
func (u *UnboundTransform) Bind(schema *Schema, caseSensitive bool) (BoundTerm, error)
func (*UnboundTransform) Equals ¶ added in v0.7.0
func (u *UnboundTransform) Equals(other UnboundTerm) bool
func (*UnboundTransform) MarshalJSON ¶ added in v0.7.0
func (u *UnboundTransform) MarshalJSON() ([]byte, error)
func (*UnboundTransform) String ¶ added in v0.7.0
func (u *UnboundTransform) String() string
func (*UnboundTransform) Transform ¶ added in v0.7.0
func (u *UnboundTransform) Transform() Transform
Transform returns the transform applied to the term.
type UnknownTransform ¶ added in v0.7.0
type UnknownTransform struct {
// contains filtered or unexported fields
}
UnknownTransform is a placeholder for a partition or sort transform that this implementation doesn't recognize. The v3 spec requires readers to load tables that use unknown transforms and to ignore those fields when filtering; writers must not commit a partition spec that uses one.
func (UnknownTransform) Apply ¶ added in v0.7.0
func (UnknownTransform) Apply(Optional[Literal]) Optional[Literal]
Apply can't be evaluated for an unknown transform.
func (UnknownTransform) CanTransform ¶ added in v0.7.0
func (UnknownTransform) CanTransform(Type) bool
CanTransform always returns true: compatibility with a source type is unverifiable for an unknown transform, not verified. Callers such as SortOrder.CheckCompatibility therefore accept any source type here.
func (UnknownTransform) Equals ¶ added in v0.7.0
func (t UnknownTransform) Equals(other Transform) bool
func (UnknownTransform) MarshalText ¶ added in v0.7.0
func (t UnknownTransform) MarshalText() ([]byte, error)
MarshalText rejects the zero value. UnknownTransform{} is a legal composite literal outside this package, and an empty name would serialize to "transform": "" -- metadata that can't be read back.
func (UnknownTransform) PreservesOrder ¶ added in v0.7.0
func (UnknownTransform) PreservesOrder() bool
func (UnknownTransform) Project ¶ added in v0.7.0
func (UnknownTransform) Project(string, BoundPredicate) (UnboundPredicate, error)
Project returns nil so scans don't prune on an unknown partition field.
func (UnknownTransform) ResultType ¶ added in v0.7.0
func (UnknownTransform) ResultType(Type) Type
ResultType is unknown, so report string, matching the Java reference.
func (UnknownTransform) String ¶ added in v0.7.0
func (t UnknownTransform) String() string
func (UnknownTransform) ToHumanStr ¶ added in v0.7.0
func (UnknownTransform) ToHumanStr(val any) string
ToHumanStr renders the value, not the transform name. Java's Transform#toHumanString default does the same and UnknownTransform doesn't override it. Returning the name would be constant for a spec field, which collapses distinct partitions into one PartitionToPath key.
func (UnknownTransform) ToHumanStrType ¶ added in v0.7.0
func (UnknownTransform) ToHumanStrType(typ Type, val any) string
type UnknownType ¶ added in v0.5.0
type UnknownType struct{}
func (UnknownType) Equals ¶ added in v0.5.0
func (UnknownType) Equals(other Type) bool
func (UnknownType) String ¶ added in v0.5.0
func (UnknownType) String() string
func (UnknownType) Type ¶ added in v0.5.0
func (UnknownType) Type() string
type VariantExtractColumn ¶ added in v0.7.0
type VariantExtractColumn struct {
Term BoundExtract
FieldID int
Name string
SourcePath []string
}
VariantExtractColumn describes a synthetic column that materializes one variant extract term for filtering.
type VariantLiteral ¶ added in v0.6.0
func (VariantLiteral) Any ¶ added in v0.6.0
func (v VariantLiteral) Any() any
func (VariantLiteral) Comparator ¶ added in v0.6.0
func (VariantLiteral) Comparator() Comparator[variant.Value]
func (VariantLiteral) Equals ¶ added in v0.6.0
func (v VariantLiteral) Equals(other Literal) bool
func (VariantLiteral) MarshalBinary ¶ added in v0.6.0
func (v VariantLiteral) MarshalBinary() ([]byte, error)
func (VariantLiteral) String ¶ added in v0.6.0
func (v VariantLiteral) String() string
func (VariantLiteral) Type ¶ added in v0.6.0
func (VariantLiteral) Type() Type
func (VariantLiteral) Value ¶ added in v0.6.0
func (v VariantLiteral) Value() variant.Value
type VariantType ¶ added in v0.6.0
type VariantType struct{}
VariantType represents semi-structured data stored using the Parquet Variant binary encoding. Requires Iceberg format version 3+.
func (VariantType) Equals ¶ added in v0.6.0
func (VariantType) Equals(other Type) bool
func (VariantType) String ¶ added in v0.6.0
func (VariantType) String() string
func (VariantType) Type ¶ added in v0.6.0
func (VariantType) Type() string
type VoidTransform ¶
type VoidTransform struct{}
VoidTransform is a transformation that always returns nil.
func (VoidTransform) CanTransform ¶ added in v0.4.0
func (VoidTransform) CanTransform(Type) bool
func (VoidTransform) Equals ¶
func (VoidTransform) Equals(other Transform) bool
func (VoidTransform) MarshalText ¶
func (t VoidTransform) MarshalText() ([]byte, error)
func (VoidTransform) PreservesOrder ¶ added in v0.2.0
func (VoidTransform) PreservesOrder() bool
func (VoidTransform) Project ¶
func (VoidTransform) Project(string, BoundPredicate) (UnboundPredicate, error)
func (VoidTransform) ResultType ¶
func (VoidTransform) ResultType(t Type) Type
func (VoidTransform) String ¶
func (VoidTransform) String() string
func (VoidTransform) ToHumanStr ¶ added in v0.2.0
func (VoidTransform) ToHumanStr(any) string
func (VoidTransform) ToHumanStrType ¶ added in v0.7.0
func (VoidTransform) ToHumanStrType(Type, any) string
type YearTransform ¶
type YearTransform struct{}
YearTransform transforms a datetime value into a year value.
func (YearTransform) Apply ¶
func (YearTransform) Apply(value Optional[Literal]) (out Optional[Literal])
func (YearTransform) CanTransform ¶ added in v0.4.0
func (t YearTransform) CanTransform(sourceType Type) bool
func (YearTransform) Equals ¶
func (YearTransform) Equals(other Transform) bool
func (YearTransform) MarshalText ¶
func (t YearTransform) MarshalText() ([]byte, error)
func (YearTransform) PreservesOrder ¶ added in v0.2.0
func (YearTransform) PreservesOrder() bool
func (YearTransform) Project ¶
func (t YearTransform) Project(name string, pred BoundPredicate) (UnboundPredicate, error)
func (YearTransform) ResultType ¶
func (YearTransform) ResultType(Type) Type
func (YearTransform) String ¶
func (YearTransform) String() string
func (YearTransform) ToHumanStr ¶ added in v0.2.0
func (YearTransform) ToHumanStr(val any) string
func (YearTransform) ToHumanStrType ¶ added in v0.7.0
func (t YearTransform) ToHumanStrType(_ Type, val any) string
func (YearTransform) Transformer ¶
Source Files
¶
- bound_set_literal_view.go
- data_file_codec.go
- data_file_refs.go
- defaults.go
- environment_context.go
- errors.go
- expr_json.go
- exprs.go
- labels.go
- literals.go
- manifest.go
- manifest_file_ref.go
- manifest_projection.go
- metadata_columns.go
- name_mapping.go
- operation_string.go
- partitions.go
- predicates.go
- schema.go
- schema_conversions.go
- transforms.go
- types.go
- utils.go
- variant_cast.go
- variant_extract.go
- variant_path.go
- visitors.go
Directories
¶
| Path | Synopsis |
|---|---|
|
Package catalog provides an interface for Catalog implementations along with a registry for registering catalog implementations.
|
Package catalog provides an interface for Catalog implementations along with a registry for registering catalog implementations. |
|
catalogtest
Package catalogtest provides a conformance suite that every catalog.Catalog implementation can run against itself, so that shared behavior is specified once rather than re-derived in each implementation's tests.
|
Package catalogtest provides a conformance suite that every catalog.Catalog implementation can run against itself, so that shared behavior is specified once rather than re-derived in each implementation's tests. |
|
hadoop
Package hadoop implements a catalog over a Hadoop-style warehouse directory.
|
Package hadoop implements a catalog over a Hadoop-style warehouse directory. |
|
rest/internal/planfake
Package planfake provides a deterministic in-process REST scan-planning server for tests.
|
Package planfake provides a deterministic in-process REST scan-planning server for tests. |
|
cmd
|
|
|
iceberg
command
|
|
|
Package codec encodes and decodes iceberg-go values for cross-process transport.
|
Package codec encodes and decodes iceberg-go values for cross-process transport. |
|
Package encryption defines interfaces and utilities for Iceberg table encryption.
|
Package encryption defines interfaces and utilities for Iceberg table encryption. |
|
datafileavro
Package datafileavro is an internal bridge that lets the github.com/apache/iceberg-go/codec package consume the iceberg package's manifest-entry Avro decoder without iceberg having to export it publicly.
|
Package datafileavro is an internal bridge that lets the github.com/apache/iceberg-go/codec package consume the iceberg package's manifest-entry Avro decoder without iceberg having to export it publicly. |
|
scanmetrics
Package scanmetrics instruments client-side scan planning.
|
Package scanmetrics instruments client-side scan planning. |
|
Package io provides an interface for IO implementations along with a registry for registering IO implementations for different URI schemes.
|
Package io provides an interface for IO implementations along with a registry for registering IO implementations for different URI schemes. |
|
gocloud
Package gocloud registers every gocloud.dev-backed FileIO implementation and therefore links the AWS, Google Cloud and Azure SDKs.
|
Package gocloud registers every gocloud.dev-backed FileIO implementation and therefore links the AWS, Google Cloud and Azure SDKs. |
|
gocloud/azure
Package azure provides the FileIO backend for Azure Data Lake Storage and Blob Storage.
|
Package azure provides the FileIO backend for Azure Data Lake Storage and Blob Storage. |
|
gocloud/blobfs
Package blobfs implements the iceberg-go/io FileIO interfaces on top of a gocloud.dev blob bucket.
|
Package blobfs implements the iceberg-go/io FileIO interfaces on top of a gocloud.dev blob bucket. |
|
gocloud/gcs
Package gcs provides the FileIO backend for Google Cloud Storage.
|
Package gcs provides the FileIO backend for Google Cloud Storage. |
|
gocloud/s3
Package s3 provides the FileIO backend for S3 and S3-compatible object stores.
|
Package s3 provides the FileIO backend for S3 and S3-compatible object stores. |
|
Package metrics implements Iceberg's Metrics Reporting API for iceberg-go.
|
Package metrics implements Iceberg's Metrics Reporting API for iceberg-go. |
|
otel
Package otel provides an OpenTelemetry-backed metrics.Reporter for iceberg-go.
|
Package otel provides an OpenTelemetry-backed metrics.Reporter for iceberg-go. |
|
Licensed to the Apache Software Foundation (ASF) under one or more contributor license agreements.
|
Licensed to the Apache Software Foundation (ASF) under one or more contributor license agreements. |
|
compaction
Package compaction provides bin-pack compaction planning for Iceberg tables.
|
Package compaction provides bin-pack compaction planning for Iceberg tables. |
|
Package udf provides the metadata model for Iceberg SQL UDFs as specified by the Iceberg UDF spec.
|
Package udf provides the metadata model for Iceberg SQL UDFs as specified by the Iceberg UDF spec. |
|
website
|
|
|
gen
command
Command gen scans the repository for example_*_test.go files that carry an "iceberg:doc" metadata block and regenerates the Concepts section of the mdbook website.
|
Command gen scans the repository for example_*_test.go files that carry an "iceberg:doc" metadata block and regenerates the Concepts section of the mdbook website. |