///|
/// Errors raised when a collection schema is malformed.
///
/// The message text is the observable behaviour of this module, so it is
/// phrased after the pymilvus wording where the two check the same thing.
pub(all) suberror SchemaError {
  SchemaError(String)
} derive(Eq, Debug)

///|
/// Key of the `dim` entry written into `type_params`.
pub let dim_key : String = "dim"

///|
/// Key of the `max_length` entry written into `type_params`.
pub let max_length_key : String = "max_length"

///|
/// One field of a collection schema.
///
/// `dim` and `max_length` are not `FieldSchema` fields in `schema.proto`;
/// the server reads them out of `type_params`, which is why they are kept
/// here as keys plus an escape hatch (`type_params`) for the remaining
/// entries such as `enable_analyzer`.
pub(all) struct Field {
  name : String
  data_type : DataType
  field_id : Int64
  is_primary_key : Bool
  auto_id : Bool
  description : String
  is_dynamic : Bool
  is_partition_key : Bool
  is_clustering_key : Bool
  nullable : Bool
  element_type : DataType?
  dim : Int?
  max_length : Int?
  type_params : Array[(String, String)]
} derive(Debug)

///|
/// A field with no optional attribute set: not a primary key, not auto ID,
/// not dynamic, not nullable, no dimension and no max length.
pub fn Field::new(name : String, data_type : DataType) -> Field {
  {
    name,
    data_type,
    field_id: 0L,
    is_primary_key: false,
    auto_id: false,
    description: "",
    is_dynamic: false,
    is_partition_key: false,
    is_clustering_key: false,
    nullable: false,
    element_type: None,
    dim: None,
    max_length: None,
    type_params: [],
  }
}

///|
/// Rejects a dimension the server would reject, so that a bad schema fails
/// locally instead of coming back as an opaque gRPC error.
fn check_dim(data_type : DataType, dim : Int) -> Unit raise SchemaError {
  if dim <= 0 {
    raise SchemaError("dimension must be positive, got \{dim}")
  }
  match data_type {
    FloatVector | Float16Vector | BFloat16Vector | Int8Vector =>
      if dim <= 1 || dim > 32768 {
        raise SchemaError("dimension must be in (1, 32768], got \{dim}")
      }
    BinaryVector =>
      if dim % 8 != 0 {
        raise SchemaError(
          "binary vector dimension must be a multiple of 8, got \{dim}",
        )
      }
    _ => ()
  }
}

///|
/// Sets the dimension of a vector field.
pub fn Field::with_dim(self : Field, dim : Int) -> Field raise SchemaError {
  check_dim(self.data_type, dim)
  { ..self, dim: Some(dim), }
}

///|
/// Sets the maximum length of a `VarChar` (or `Text`) field.
pub fn Field::with_max_length(
  self : Field,
  max_length : Int,
) -> Field raise SchemaError {
  if max_length <= 0 {
    raise SchemaError("max_length must be positive, got \{max_length}")
  }
  if self.data_type != VarChar &&
    self.data_type != String &&
    self.data_type != Text {
    raise SchemaError(
      "max_length is only valid for VarChar, Text and String fields, got \{self.data_type.name()}",
    )
  }
  { ..self, max_length: Some(max_length), }
}

///|
/// Adds an arbitrary `type_params` entry, e.g. `enable_analyzer=true`.
pub fn Field::with_type_param(
  self : Field,
  key : String,
  value : String,
) -> Field {
  { ..self, type_params: self.type_params + [(key, value)], }
}

///|
/// Marks the field as the collection's primary key.
pub fn Field::as_primary_key(self : Field) -> Field {
  { ..self, is_primary_key: true, }
}

///|
/// Enables server-side ID generation for a primary key field.
pub fn Field::as_auto_id(self : Field) -> Field {
  { ..self, auto_id: true, }
}

///|
/// Allows null values for the field.
pub fn Field::as_nullable(self : Field) -> Field {
  { ..self, nullable: true, }
}

///|
/// Enables logic partitions on the field.
pub fn Field::as_partition_key(self : Field) -> Field {
  { ..self, is_partition_key: true, }
}

///|
/// Sets the element type of an `Array` field.
pub fn Field::with_element_type(self : Field, element_type : DataType) -> Field {
  { ..self, element_type: Some(element_type), }
}

///|
/// The `type_params` entries `schema.proto` expects for this field: the
/// dimension for vectors, the max length for string fields.
pub fn Field::type_params_pairs(self : Field) -> Array[(String, String)] {
  let pairs = []
  match self.dim {
    Some(dim) => pairs.push((dim_key, dim.to_string()))
    None => ()
  }
  match self.max_length {
    Some(max_length) => pairs.push((max_length_key, max_length.to_string()))
    None => ()
  }
  pairs + self.type_params
}

///|
/// Whether the field is a sparse float vector field, which is the one vector
/// kind that has no dimension.
pub fn Field::is_sparse_vector(self : Field) -> Bool {
  self.data_type == SparseFloatVector
}

///|
/// Validates one field: a dimension on every dense vector field and on none
/// of the others, an auto ID only on the primary key, and a partition key
/// only on an `Int64` or `VarChar` field.
pub fn Field::validate(self : Field) -> Unit raise SchemaError {
  if self.data_type.is_dense_vector_type() {
    match self.dim {
      Some(_) => ()
      None =>
        raise SchemaError("vector field \{self.name} requires a dimension")
    }
  } else {
    match self.dim {
      Some(_) =>
        if self.data_type != SparseFloatVector &&
          self.data_type != ArrayOfVector {
          raise SchemaError(
            "dimension is only valid for vector fields, but \{self.name} is \{self.data_type.name()}",
          )
        }
      None => ()
    }
  }
  if self.auto_id && !self.is_primary_key {
    raise SchemaError(
      "auto_id is only valid on the primary key field \{self.name}",
    )
  }
  if self.is_partition_key &&
    self.data_type != Int64 &&
    self.data_type != VarChar {
    raise SchemaError(
      "partition key field \{self.name} must be Int64 or VarChar",
    )
  }
}

///|
/// A collection schema: the fields plus the collection-level description and
/// the `enable_dynamic_field` switch from `CollectionSchema` in
/// `schema.proto`.
pub(all) struct CollectionSchema {
  fields : Array[Field]
  description : String
  enable_dynamic_field : Bool
} derive(Debug)

///|
/// A schema with the given fields, no dynamic field.
pub fn CollectionSchema::new(fields : Array[Field]) -> CollectionSchema {
  { fields, description: "", enable_dynamic_field: false, }
}

///|
/// A schema with a single field, for the common one-field case.
pub fn CollectionSchema::from_field(field : Field) -> CollectionSchema {
  { fields: [field], description: "", enable_dynamic_field: false, }
}

///|
/// Sets the collection description.
pub fn CollectionSchema::with_description(
  self : CollectionSchema,
  description : String,
) -> CollectionSchema {
  { ..self, description, }
}

///|
/// Enables or disables the `$meta` dynamic field.
pub fn CollectionSchema::with_dynamic_field(
  self : CollectionSchema,
  enabled : Bool,
) -> CollectionSchema {
  { ..self, enable_dynamic_field: enabled, }
}

///|
/// Appends a field.
pub fn CollectionSchema::with_field(
  self : CollectionSchema,
  field : Field,
) -> CollectionSchema {
  { ..self, fields: self.fields + [field], }
}

///|
/// The field with the given name, if any.
pub fn CollectionSchema::field(
  self : CollectionSchema,
  name : String,
) -> Field? {
  for field in self.fields {
    if field.name == name {
      return Some(field)
    }
  }
  None
}

///|
/// The primary key field, if the schema declares one.
pub fn CollectionSchema::primary_key(self : CollectionSchema) -> Field? {
  for field in self.fields {
    if field.is_primary_key {
      return Some(field)
    }
  }
  None
}

///|
/// All vector fields, in declaration order.
pub fn CollectionSchema::vector_fields(self : CollectionSchema) -> Array[Field] {
  self.fields.filter(field => field.data_type.is_vector_type())
}

///|
/// Validates the whole schema: at least one field, exactly one primary key of
/// an auto-ID-capable type, no duplicate names, and every field individually
/// valid.
pub fn CollectionSchema::validate(
  self : CollectionSchema,
) -> Unit raise SchemaError {
  if self.fields.is_empty() {
    raise SchemaError("schema must have at least one field")
  }
  let mut primary_keys = 0
  for field in self.fields {
    if field.is_primary_key {
      primary_keys = primary_keys + 1
    }
  }
  if primary_keys != 1 {
    raise SchemaError(
      "schema must have exactly one primary key field, got \{primary_keys}",
    )
  }
  let seen : Map[String, Unit] = Map([])
  for field in self.fields {
    if seen.contains(field.name) {
      raise SchemaError("duplicate field name: \{field.name}")
    }
    seen[field.name] = ()
    field.validate()
  }
  match self.primary_key() {
    Some(pk) =>
      match pk.data_type {
        Int64 => ()
        VarChar =>
          if pk.auto_id {
            raise SchemaError(
              "auto_id is not supported for VarChar primary keys",
            )
          }
        _ =>
          raise SchemaError(
            "primary key field \{pk.name} must be Int64 or VarChar, got \{pk.data_type.name()}",
          )
      }
    None => ()
  }
}

///|
/// A schema with one auto ID `Int64` primary key named `id`, plus the given
/// fields. This is the shape most examples use.
pub fn CollectionSchema::with_auto_id_primary(
  fields : Array[Field],
) -> CollectionSchema {
  let id = Field::new("id", Int64).as_primary_key().as_auto_id()
  CollectionSchema::new([id] + fields)
}

///|

// `derive(Debug)` (and `derive(Eq)`) promote their trait methods to plain
// methods, which MoonBit deprecates. Pinning the promotions here keeps the
// module warning-free without dropping the derives themselves.

///|
#deprecated
pub extend SchemaError with Eq::{not_equal, equal}

///|
#deprecated
pub extend SchemaError with @debug.Debug::{to_repr}

///|
#deprecated
pub extend Field with @debug.Debug::{to_repr}

///|
#deprecated
pub extend CollectionSchema with @debug.Debug::{to_repr}