Files
seaweedfs/weed/s3api/iceberg/iceberg_create_view_test.go
T
Chris Lu 28862c866e Authorize an Iceberg table create before it writes (#10991)
* s3tables: share one CreateTable authorization gate

CreateTable and RegisterTable each carried their own copy of the name
validation, policy load and permission check. Fold them into
authorizeCreateTable, and expose it on the Manager for callers that write
into a table bucket before the table itself is registered.

Claude-Session: https://claude.ai/code/session_01QiJkka1T2NAWDWq4JQ8Vuy

* iceberg: authorize a table create before it writes

Stage-create returns before the S3Tables registration that authorizes a
create, and the plain create writes its metadata file before reaching it,
so a caller who may not create the table could still leave a staged
template, a marker and a v1.metadata.json in the target bucket - and get
vended credentials for a location of their choosing. Run the CreateTable
gate as soon as the table is known to be absent.

Claude-Session: https://claude.ai/code/session_01QiJkka1T2NAWDWq4JQ8Vuy

* iceberg: authorize a create-on-commit the same way

A commit against a table that does not exist creates it, writing the
metadata file first and only then reaching the registration that checks
the caller may create it. Denied callers saw a 500 for what is a 403.

Claude-Session: https://claude.ai/code/session_01QiJkka1T2NAWDWq4JQ8Vuy

* iceberg: pin that identity actions reach the create gate

The manager request is built from the caller's own context, so an identity
whose actions carry the permission still passes. Worth a test: a fresh
context here would silently deny every such caller.

Claude-Session: https://claude.ai/code/session_01QiJkka1T2NAWDWq4JQ8Vuy
2026-08-27 16:23:51 -07:00

234 lines
7.9 KiB
Go

package iceberg
import (
"bytes"
"context"
"encoding/json"
"errors"
"io"
"net/http"
"net/http/httptest"
"path"
"sort"
"strings"
"testing"
"github.com/apache/iceberg-go"
"github.com/apache/iceberg-go/table"
"github.com/apache/iceberg-go/view"
"github.com/gorilla/mux"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3tables"
"google.golang.org/grpc"
)
// memFiler is a minimal in-memory SeaweedFilerClient backing the view handlers
// end-to-end through the s3tables manager. Only the entry operations the create
// path touches are implemented; anything else panics so a missing dependency is
// loud rather than silently passing.
type memFiler struct {
filer_pb.SeaweedFilerClient
entries map[string]*filer_pb.Entry
// failFileCreate, when set, makes CreateEntry fail for non-directory entries,
// simulating a metadata-file write error.
failFileCreate error
}
func newMemFiler() *memFiler {
return &memFiler{entries: map[string]*filer_pb.Entry{}}
}
func (m *memFiler) WithFilerClient(_ bool, fn func(client filer_pb.SeaweedFilerClient) error) error {
return fn(m)
}
func (m *memFiler) seed(p string, entry *filer_pb.Entry) {
m.entries[p] = entry
}
func (m *memFiler) LookupDirectoryEntry(_ context.Context, in *filer_pb.LookupDirectoryEntryRequest, _ ...grpc.CallOption) (*filer_pb.LookupDirectoryEntryResponse, error) {
entry, ok := m.entries[path.Join(in.Directory, in.Name)]
if !ok {
return nil, filer_pb.ErrNotFound
}
return &filer_pb.LookupDirectoryEntryResponse{Entry: entry}, nil
}
func (m *memFiler) CreateEntry(_ context.Context, in *filer_pb.CreateEntryRequest, _ ...grpc.CallOption) (*filer_pb.CreateEntryResponse, error) {
if m.failFileCreate != nil && !in.Entry.IsDirectory {
return nil, m.failFileCreate
}
m.entries[path.Join(in.Directory, in.Entry.Name)] = in.Entry
return &filer_pb.CreateEntryResponse{}, nil
}
func (m *memFiler) DeleteEntry(_ context.Context, in *filer_pb.DeleteEntryRequest, _ ...grpc.CallOption) (*filer_pb.DeleteEntryResponse, error) {
prefix := path.Join(in.Directory, in.Name)
for p := range m.entries {
if p == prefix || strings.HasPrefix(p, prefix+"/") {
delete(m.entries, p)
}
}
return &filer_pb.DeleteEntryResponse{}, nil
}
func (m *memFiler) UpdateEntry(_ context.Context, in *filer_pb.UpdateEntryRequest, _ ...grpc.CallOption) (*filer_pb.UpdateEntryResponse, error) {
m.entries[path.Join(in.Directory, in.Entry.Name)] = in.Entry
return &filer_pb.UpdateEntryResponse{}, nil
}
func (m *memFiler) ListEntries(_ context.Context, in *filer_pb.ListEntriesRequest, _ ...grpc.CallOption) (filer_pb.SeaweedFiler_ListEntriesClient, error) {
var names []string
for p := range m.entries {
if path.Dir(p) == in.Directory {
names = append(names, p)
}
}
sort.Strings(names)
stream := &memListEntries{}
for _, name := range names {
stream.entries = append(stream.entries, m.entries[name])
}
return stream, nil
}
type memListEntries struct {
grpc.ClientStream
entries []*filer_pb.Entry
}
func (m *memListEntries) Recv() (*filer_pb.ListEntriesResponse, error) {
if len(m.entries) == 0 {
return nil, io.EOF
}
entry := m.entries[0]
m.entries = m.entries[1:]
return &filer_pb.ListEntriesResponse{Entry: entry}, nil
}
func newCreateViewRequest(t *testing.T, namespace, name, sql string) *http.Request {
t.Helper()
schema := iceberg.NewSchemaWithIdentifiers(0, nil,
iceberg.NestedField{ID: 1, Name: "id", Type: iceberg.PrimitiveTypes.Int32, Required: true},
)
ver, err := view.NewVersionFromSQL(1, schema.ID, sql, table.Identifier{namespace})
if err != nil {
t.Fatal(err)
}
body, err := json.Marshal(CreateViewRequest{Name: name, Schema: schema, ViewVersion: ver})
if err != nil {
t.Fatal(err)
}
r := httptest.NewRequest(http.MethodPost, "/v1/namespaces/"+namespace+"/views", bytes.NewReader(body))
r = mux.SetURLVars(r, map[string]string{"namespace": namespace})
r = r.WithContext(s3_constants.SetIdentityNameInContext(r.Context(), s3_constants.AccountAdminId))
return r
}
// seedNamespace registers a bucket and namespace owned by owner so the s3tables
// existence and auth-context lookups pass.
func seedNamespace(fc *memFiler, bucket, namespace, owner string) {
fc.seed(s3tables.GetTableBucketPath(bucket), &filer_pb.Entry{Name: bucket, IsDirectory: true})
meta, _ := json.Marshal(map[string]any{"namespace": []string{namespace}, "ownerAccountId": owner})
fc.seed(s3tables.GetNamespacePath(bucket, namespace), &filer_pb.Entry{
Name: namespace,
IsDirectory: true,
Extended: map[string][]byte{s3tables.ExtendedKeyMetadata: meta},
})
}
func TestCreateViewMissingNamespaceReturns404(t *testing.T) {
fc := newMemFiler()
s := NewServer(fc, nil)
w := httptest.NewRecorder()
s.handleCreateView(w, newCreateViewRequest(t, "ns", "v", "SELECT 1"))
if w.Code != http.StatusNotFound {
t.Fatalf("status = %d, want %d (body: %s)", w.Code, http.StatusNotFound, w.Body.String())
}
var errResp ErrorResponse
if err := json.Unmarshal(w.Body.Bytes(), &errResp); err != nil {
t.Fatalf("unmarshal error body: %v", err)
}
if errResp.Error.Type != "NoSuchNamespaceException" {
t.Fatalf("error type = %q, want NoSuchNamespaceException", errResp.Error.Type)
}
}
func TestCreateViewTagsEntryAsView(t *testing.T) {
const bucket = "warehouse"
fc := newMemFiler()
seedNamespace(fc, bucket, "ns", s3_constants.AccountAdminId)
s := NewServer(fc, nil)
w := httptest.NewRecorder()
s.handleCreateView(w, newCreateViewRequest(t, "ns", "v", "SELECT 1"))
if w.Code != http.StatusOK {
t.Fatalf("create status = %d, want 200 (body: %s)", w.Code, w.Body.String())
}
entry, ok := fc.entries[s3tables.GetTablePath(bucket, "ns", "v")]
if !ok {
t.Fatalf("view entry not created")
}
if got := string(entry.Extended[s3tables.ExtendedKeyEntryType]); got != s3tables.EntryTypeView {
t.Fatalf("entryType = %q, want %q", got, s3tables.EntryTypeView)
}
if _, ok := entry.Extended[s3tables.ExtendedKeyMetadata]; !ok {
t.Fatalf("view entry missing metadata attribute")
}
}
func TestCreateViewDuplicateDoesNotClobberMetadata(t *testing.T) {
const bucket = "warehouse"
fc := newMemFiler()
seedNamespace(fc, bucket, "ns", s3_constants.AccountAdminId)
s := NewServer(fc, nil)
w := httptest.NewRecorder()
s.handleCreateView(w, newCreateViewRequest(t, "ns", "v", "SELECT 1"))
if w.Code != http.StatusOK {
t.Fatalf("first create status = %d, want 200 (body: %s)", w.Code, w.Body.String())
}
metadataKey := path.Join(s3tables.GetTablePath(bucket, "ns", "v"), "metadata", "v1.metadata.json")
first, ok := fc.entries[metadataKey]
if !ok {
t.Fatalf("metadata file not written at %s", metadataKey)
}
original := append([]byte(nil), first.Content...)
// Re-create the same view with different SQL: the existence pre-check must
// short-circuit so the persisted v1.metadata.json stays byte-for-byte intact.
w = httptest.NewRecorder()
s.handleCreateView(w, newCreateViewRequest(t, "ns", "v", "SELECT 2"))
if w.Code != http.StatusOK {
t.Fatalf("duplicate create status = %d, want 200 (body: %s)", w.Code, w.Body.String())
}
if got := fc.entries[metadataKey].Content; !bytes.Equal(got, original) {
t.Fatalf("duplicate create clobbered stored metadata:\n got %s\nwant %s", got, original)
}
}
func TestCreateViewRollsBackEntryWhenMetadataWriteFails(t *testing.T) {
const bucket = "warehouse"
fc := newMemFiler()
seedNamespace(fc, bucket, "ns", s3_constants.AccountAdminId)
fc.failFileCreate = errors.New("disk full")
s := NewServer(fc, nil)
w := httptest.NewRecorder()
s.handleCreateView(w, newCreateViewRequest(t, "ns", "v", "SELECT 1"))
if w.Code != http.StatusInternalServerError {
t.Fatalf("status = %d, want %d (body: %s)", w.Code, http.StatusInternalServerError, w.Body.String())
}
// The registered view must be rolled back so it doesn't linger pointing at
// metadata that was never written.
if _, ok := fc.entries[s3tables.GetTablePath(bucket, "ns", "v")]; ok {
t.Fatalf("view entry left behind after metadata write failure")
}
}